@mrciphersmith/keryx 0.2.164 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. package/dist/cli.js +85355 -56749
  2. package/dist/core.js +28605 -18901
  3. package/package.json +2 -2
  4. package/src/gdgraph/affected-report.ts +141 -0
  5. package/src/gdgraph/build.ts +170 -23
  6. package/src/gdgraph/service.ts +6 -0
  7. package/src/gdgraph/staleness.ts +253 -45
  8. package/src/gdskills/bundled/agents/codebase-navigator.md +55 -0
  9. package/src/gdskills/bundled/agents/design-advisor.md +64 -0
  10. package/src/gdskills/bundled/agents/docs-maintainer.md +56 -0
  11. package/src/gdskills/bundled/agents/end-to-end-tester.md +56 -0
  12. package/src/gdskills/bundled/agents/error-path-auditor.md +57 -0
  13. package/src/gdskills/bundled/agents/go-build-fixer.md +52 -0
  14. package/src/gdskills/bundled/agents/go-code-auditor.md +49 -0
  15. package/src/gdskills/bundled/agents/performance-auditor.md +63 -0
  16. package/src/gdskills/bundled/agents/python-build-fixer.md +52 -0
  17. package/src/gdskills/bundled/agents/python-code-auditor.md +49 -0
  18. package/src/gdskills/bundled/agents/refactoring-steward.md +61 -0
  19. package/src/gdskills/bundled/agents/security-auditor.md +62 -0
  20. package/src/gdskills/bundled/agents/test-first-driver.md +61 -0
  21. package/src/gdskills/bundled/agents/work-planner.md +62 -0
  22. package/src/gdskills/bundled/install-manifest.json +530 -0
  23. package/src/gdskills/bundled/rules/core/skill-lifecycle.mdc +29 -1
  24. package/src/gdskills/bundled/rules/core/skills-storage-workflow.mdc +2 -2
  25. package/src/gdskills/bundled/skills/review/code-style-review/SKILL.md +1 -1
  26. package/src/gdskills/bundled/skills/review/review-orchestrator/SKILL.md +74 -246
  27. package/src/gdskills/bundled/skills/review/review-orchestrator/output-contract.schema.json +19 -0
  28. package/src/gdskills/bundled/skills/review/review-orchestrator/reviewer-finding.schema.json +10 -0
  29. package/src/gdskills/bundled/skills/review/review-orchestrator/reviewer-input.schema.json +5 -0
  30. package/src/gdskills/bundled/skills/review/review-orchestrator/templates/pr-comment-backend.md +50 -0
  31. package/src/gdskills/bundled/skills/review/review-orchestrator/templates/pr-comment-frontend.md +52 -0
  32. package/src/gdskills/bundled/skills/review/review-orchestrator/templates/review-report.md +143 -0
  33. package/src/gdskills/bundled/stacks/go/agent-refs.json +3 -0
  34. package/src/gdskills/bundled/stacks/go/governance/eval.json +1745 -0
  35. package/src/gdskills/bundled/stacks/go/governance/scout.json +31 -0
  36. package/src/gdskills/bundled/stacks/go/pack.json +41 -0
  37. package/src/gdskills/bundled/stacks/go/rules/coding-style.mdc +85 -0
  38. package/src/gdskills/bundled/stacks/go/rules/patterns.mdc +65 -0
  39. package/src/gdskills/bundled/stacks/go/rules/security.mdc +73 -0
  40. package/src/gdskills/bundled/stacks/go/rules/testing.mdc +68 -0
  41. package/src/gdskills/bundled/stacks/go/skills/go-build-fix/SKILL.md +138 -0
  42. package/src/gdskills/bundled/stacks/go/skills/go-build-fix/evals.json +75 -0
  43. package/src/gdskills/bundled/stacks/go/skills/go-code-review/SKILL.md +121 -0
  44. package/src/gdskills/bundled/stacks/go/skills/go-code-review/evals.json +72 -0
  45. package/src/gdskills/bundled/stacks/go/skills/go-implementation/SKILL.md +122 -0
  46. package/src/gdskills/bundled/stacks/go/skills/go-implementation/evals.json +76 -0
  47. package/src/gdskills/bundled/stacks/go/skills/go-testing/SKILL.md +126 -0
  48. package/src/gdskills/bundled/stacks/go/skills/go-testing/evals.json +73 -0
  49. package/src/gdskills/bundled/stacks/python/agent-refs.json +3 -0
  50. package/src/gdskills/bundled/stacks/python/governance/eval.json +1758 -0
  51. package/src/gdskills/bundled/stacks/python/governance/scout.json +34 -0
  52. package/src/gdskills/bundled/stacks/python/pack.json +41 -0
  53. package/src/gdskills/bundled/stacks/python/rules/coding-style.mdc +63 -0
  54. package/src/gdskills/bundled/stacks/python/rules/patterns.mdc +88 -0
  55. package/src/gdskills/bundled/stacks/python/rules/security.mdc +84 -0
  56. package/src/gdskills/bundled/stacks/python/rules/testing.mdc +77 -0
  57. package/src/gdskills/bundled/stacks/python/skills/python-build-fix/SKILL.md +144 -0
  58. package/src/gdskills/bundled/stacks/python/skills/python-build-fix/evals.json +74 -0
  59. package/src/gdskills/bundled/stacks/python/skills/python-code-review/SKILL.md +155 -0
  60. package/src/gdskills/bundled/stacks/python/skills/python-code-review/evals.json +72 -0
  61. package/src/gdskills/bundled/stacks/python/skills/python-implementation/SKILL.md +143 -0
  62. package/src/gdskills/bundled/stacks/python/skills/python-implementation/evals.json +78 -0
  63. package/src/gdskills/bundled/stacks/python/skills/python-testing/SKILL.md +132 -0
  64. package/src/gdskills/bundled/stacks/python/skills/python-testing/evals.json +73 -0
  65. package/src/gdskills/bundled/stacks/react/agent-refs.json +4 -0
  66. package/src/gdskills/bundled/stacks/react/governance/eval.json +2188 -0
  67. package/src/gdskills/bundled/stacks/react/governance/scout.json +40 -0
  68. package/src/gdskills/bundled/stacks/react/pack.json +42 -0
  69. package/src/gdskills/bundled/stacks/react/rules/coding-style.mdc +58 -0
  70. package/src/gdskills/bundled/stacks/react/rules/patterns.mdc +79 -0
  71. package/src/gdskills/bundled/stacks/react/rules/security.mdc +70 -0
  72. package/src/gdskills/bundled/stacks/react/rules/testing.mdc +60 -0
  73. package/src/gdskills/bundled/stacks/react/skills/react-build-fix/SKILL.md +139 -0
  74. package/src/gdskills/bundled/stacks/react/skills/react-build-fix/evals.json +72 -0
  75. package/src/gdskills/bundled/stacks/react/skills/react-code-review/SKILL.md +148 -0
  76. package/src/gdskills/bundled/stacks/react/skills/react-code-review/evals.json +74 -0
  77. package/src/gdskills/bundled/stacks/react/skills/react-implementation/SKILL.md +140 -0
  78. package/src/gdskills/bundled/stacks/react/skills/react-implementation/evals.json +74 -0
  79. package/src/gdskills/bundled/stacks/react/skills/react-testing/SKILL.md +142 -0
  80. package/src/gdskills/bundled/stacks/react/skills/react-testing/evals.json +83 -0
  81. package/src/gdskills/bundled/stacks/react/skills/react-upgrade-migration/SKILL.md +155 -0
  82. package/src/gdskills/bundled/stacks/react/skills/react-upgrade-migration/evals.json +74 -0
  83. package/src/gdskills/bundled/stacks/ts-js-node/agent-refs.json +4 -0
  84. package/src/gdskills/bundled/stacks/ts-js-node/governance/eval.json +2155 -0
  85. package/src/gdskills/bundled/stacks/ts-js-node/governance/scout.json +40 -0
  86. package/src/gdskills/bundled/stacks/ts-js-node/pack.json +41 -0
  87. package/src/gdskills/bundled/stacks/ts-js-node/rules/coding-style.mdc +73 -0
  88. package/src/gdskills/bundled/stacks/ts-js-node/rules/patterns.mdc +61 -0
  89. package/src/gdskills/bundled/stacks/ts-js-node/rules/security.mdc +71 -0
  90. package/src/gdskills/bundled/stacks/ts-js-node/rules/testing.mdc +63 -0
  91. package/src/gdskills/bundled/stacks/ts-js-node/skills/nodejs-build-fix/SKILL.md +137 -0
  92. package/src/gdskills/bundled/stacks/ts-js-node/skills/nodejs-build-fix/evals.json +73 -0
  93. package/src/gdskills/bundled/stacks/ts-js-node/skills/nodejs-code-review/SKILL.md +124 -0
  94. package/src/gdskills/bundled/stacks/ts-js-node/skills/nodejs-code-review/evals.json +74 -0
  95. package/src/gdskills/bundled/stacks/ts-js-node/skills/nodejs-esm-migration/SKILL.md +152 -0
  96. package/src/gdskills/bundled/stacks/ts-js-node/skills/nodejs-esm-migration/evals.json +71 -0
  97. package/src/gdskills/bundled/stacks/ts-js-node/skills/nodejs-implementation/SKILL.md +127 -0
  98. package/src/gdskills/bundled/stacks/ts-js-node/skills/nodejs-implementation/evals.json +72 -0
  99. package/src/gdskills/bundled/stacks/ts-js-node/skills/nodejs-testing/SKILL.md +134 -0
  100. package/src/gdskills/bundled/stacks/ts-js-node/skills/nodejs-testing/evals.json +70 -0
@@ -0,0 +1,1745 @@
1
+ {
2
+ "schemaVersion": "1.0.0",
3
+ "reports": [
4
+ {
5
+ "schemaVersion": "1.0.0",
6
+ "skillId": "go/go-build-fix",
7
+ "strictness": "high",
8
+ "trials": 10,
9
+ "triggerAccuracy": {
10
+ "truePositive": 6,
11
+ "falsePositive": 0,
12
+ "positives": 6,
13
+ "negatives": 6
14
+ },
15
+ "evidence": "authored",
16
+ "scenarios": [
17
+ {
18
+ "id": "trigger-positive-1",
19
+ "kind": "trigger-positive",
20
+ "prompt": "go build is failing with an undefined symbol error",
21
+ "strictness": "high",
22
+ "trials": 1,
23
+ "passes": 1,
24
+ "passRate": 1,
25
+ "passAtK": 1,
26
+ "grader": "trigger-rank-fork-family",
27
+ "status": "ran",
28
+ "deterministic": true
29
+ },
30
+ {
31
+ "id": "trigger-positive-2",
32
+ "kind": "trigger-positive",
33
+ "prompt": "go.mod and go.sum are out of sync, please fix",
34
+ "strictness": "high",
35
+ "trials": 1,
36
+ "passes": 1,
37
+ "passRate": 1,
38
+ "passAtK": 1,
39
+ "grader": "trigger-rank-fork-family",
40
+ "status": "ran",
41
+ "deterministic": true
42
+ },
43
+ {
44
+ "id": "trigger-positive-3",
45
+ "kind": "trigger-positive",
46
+ "prompt": "Resolve this Go import cycle between two internal packages",
47
+ "strictness": "high",
48
+ "trials": 1,
49
+ "passes": 1,
50
+ "passRate": 1,
51
+ "passAtK": 1,
52
+ "grader": "trigger-rank-fork-family",
53
+ "status": "ran",
54
+ "deterministic": true
55
+ },
56
+ {
57
+ "id": "trigger-positive-4",
58
+ "kind": "trigger-positive",
59
+ "prompt": "go vet is reporting a printf format error",
60
+ "strictness": "high",
61
+ "trials": 1,
62
+ "passes": 1,
63
+ "passRate": 1,
64
+ "passAtK": 1,
65
+ "grader": "trigger-rank-fork-family",
66
+ "status": "ran",
67
+ "deterministic": true
68
+ },
69
+ {
70
+ "id": "trigger-positive-5",
71
+ "kind": "trigger-positive",
72
+ "prompt": "golangci-lint is failing on this Go package",
73
+ "strictness": "high",
74
+ "trials": 1,
75
+ "passes": 1,
76
+ "passRate": 1,
77
+ "passAtK": 1,
78
+ "grader": "trigger-rank-fork-family",
79
+ "status": "ran",
80
+ "deterministic": true
81
+ },
82
+ {
83
+ "id": "trigger-positive-6",
84
+ "kind": "trigger-positive",
85
+ "prompt": "go test -race is failing with a data race, fix the build",
86
+ "strictness": "high",
87
+ "trials": 1,
88
+ "passes": 1,
89
+ "passRate": 1,
90
+ "passAtK": 1,
91
+ "grader": "trigger-rank-fork-family",
92
+ "status": "ran",
93
+ "deterministic": true
94
+ },
95
+ {
96
+ "id": "trigger-negative-1",
97
+ "kind": "trigger-negative",
98
+ "prompt": "npm install is failing with a peer dependency conflict",
99
+ "strictness": "high",
100
+ "trials": 1,
101
+ "passes": 1,
102
+ "passRate": 1,
103
+ "passAtK": 1,
104
+ "grader": "trigger-rank-fork-family",
105
+ "status": "ran",
106
+ "deterministic": true
107
+ },
108
+ {
109
+ "id": "trigger-negative-2",
110
+ "kind": "trigger-negative",
111
+ "prompt": "cargo build is failing for this Rust crate",
112
+ "strictness": "high",
113
+ "trials": 1,
114
+ "passes": 1,
115
+ "passRate": 1,
116
+ "passAtK": 1,
117
+ "grader": "trigger-rank-fork-family",
118
+ "status": "ran",
119
+ "deterministic": true
120
+ },
121
+ {
122
+ "id": "trigger-negative-3",
123
+ "kind": "trigger-negative",
124
+ "prompt": "pip install is failing for this Python project",
125
+ "strictness": "high",
126
+ "trials": 1,
127
+ "passes": 1,
128
+ "passRate": 1,
129
+ "passAtK": 1,
130
+ "grader": "trigger-rank-fork-family",
131
+ "status": "ran",
132
+ "deterministic": true
133
+ },
134
+ {
135
+ "id": "trigger-negative-4",
136
+ "kind": "trigger-negative",
137
+ "prompt": "Implement a new Go feature in the order package",
138
+ "strictness": "high",
139
+ "trials": 1,
140
+ "passes": 1,
141
+ "passRate": 1,
142
+ "passAtK": 1,
143
+ "grader": "trigger-rank-fork-family",
144
+ "status": "ran",
145
+ "deterministic": true
146
+ },
147
+ {
148
+ "id": "trigger-negative-5",
149
+ "kind": "trigger-negative",
150
+ "prompt": "Review this Go diff for concurrency bugs",
151
+ "strictness": "high",
152
+ "trials": 1,
153
+ "passes": 1,
154
+ "passRate": 1,
155
+ "passAtK": 1,
156
+ "grader": "trigger-rank-fork-family",
157
+ "status": "ran",
158
+ "deterministic": true
159
+ },
160
+ {
161
+ "id": "trigger-negative-6",
162
+ "kind": "trigger-negative",
163
+ "prompt": "Write Go tests for this function",
164
+ "strictness": "high",
165
+ "trials": 1,
166
+ "passes": 1,
167
+ "passRate": 1,
168
+ "passAtK": 1,
169
+ "grader": "trigger-rank-fork-family",
170
+ "status": "ran",
171
+ "deterministic": true
172
+ },
173
+ {
174
+ "id": "module-mismatch",
175
+ "kind": "behavior",
176
+ "prompt": "go build fails saying go.sum is out of date after I bumped a dependency in go.mod. How do I fix it?",
177
+ "strictness": "high",
178
+ "trials": 10,
179
+ "passes": 10,
180
+ "passRate": 1,
181
+ "passAtK": 1,
182
+ "grader": "judge",
183
+ "status": "ran",
184
+ "trialRecords": [
185
+ {
186
+ "output": "Run `go mod tidy` in the module root. That recomputes `go.sum` from the requirements now listed in `go.mod` (and prunes anything no longer needed), which is exactly the mismatch you're hitting.\n\nThen verify from a clean state so you know nothing stale was cached:\n\n```bash\ngo mod tidy\ngo build ./...\ngo vet ./...\ngo test -race ./...\n```\n\nA couple of things to check while you're there:\n\n- If `go mod tidy` errors instead of resolving, read the error before touching anything else \u2014 it usually names a specific module/version it can't satisfy, and that's a real version conflict, not just a stale sum.\n- Don't reach for a `replace` directive or bump the `go`/`toolchain` directive to make the error go away. A stale `go.sum` is fixed by `go mod tidy` alone; those other moves change the module's declared requirements for every consumer and should only happen for a reason you can name.\n- If only one dependency was bumped, `go mod tidy` is still the right call rather than a manual `go.sum` edit \u2014 hand-editing hashes is how you get a genuinely corrupted sum line later.\n\nIf `go build ./...` still fails after `go mod tidy`, paste the exact error text \u2014 that gets you into a different classification (import cycle, generic inference, vet finding, etc.) and a different fix.",
187
+ "outputSha256": "fdfe7b236eb0d327dff1fdf944827730dea72c5e3cb81430326c377277367445",
188
+ "promptSha256": "c2c5c9df499cd7b1cf97b0043b920e2b1f07f48e7e636973e01a285741e0307f",
189
+ "deterministic": [],
190
+ "judge": {
191
+ "verdict": "pass",
192
+ "reason": "Names `go mod tidy` in module root (\"Run `go mod tidy` in the module root\") and explains root cause: recomputes go.sum from requirements now in go.mod, \"exactly the mismatch you're hitting.\" Verification shown: `go build ./...`, `go vet`, `go test -race`. Warns against bumping go/toolchain and hand-editing go.sum hashes. No fail criteria met."
193
+ },
194
+ "passed": true
195
+ },
196
+ {
197
+ "output": "Run `go mod tidy` from the module root. That reconciles `go.sum` against `go.mod` by fetching the exact module versions and checksums your build now requires, adding missing `go.sum` entries and pruning ones no longer needed.\n\n```bash\ngo mod tidy\n```\n\nThen verify:\n\n```bash\ngo build ./...\ngo vet ./...\ngo test -race ./...\n```\n\nIf `go mod tidy` itself fails \u2014 e.g. it can't resolve a version, or a checksum conflict surfaces (`verifying module: checksum mismatch`) \u2014 that points at a different problem than a stale `go.sum`:\n\n- **Checksum mismatch**: a cached copy in `GOMODCACHE` disagrees with the recorded hash. Run `go clean -modcache` and re-run `go mod tidy` to refetch from a clean state.\n- **Version resolution failure**: run `go mod why -m <module>` and `go mod graph` to see what pulled the dependency in before touching any version by hand.\n- **A `replace` directive** pointing at a stale local path will also make `go mod tidy` fail; only fix it if it's a real, intentional override, not something to route around the error with.\n\nDon't manually edit `go.sum` \u2014 it's a generated lockfile, and hand-edits get it out of sync again on the next `go mod tidy`. Don't bump the `go` directive in `go.mod` just to make resolution succeed either; that changes the declared minimum toolchain for everyone and doesn't address why the dependency was bumped in the first place.\n\nOnce `go mod tidy` runs clean and the three verification commands exit 0, the fix is complete. Report the root cause in one line, e.g. \"go.sum predated the dependency bump in go.mod \u2014 ran `go mod tidy`.\"",
198
+ "outputSha256": "d53d79328d40d3f4343c752e308a8c6760606d0c0f7cef4775f0b0222b85f8d4",
199
+ "promptSha256": "c2c5c9df499cd7b1cf97b0043b920e2b1f07f48e7e636973e01a285741e0307f",
200
+ "deterministic": [],
201
+ "judge": {
202
+ "verdict": "pass",
203
+ "reason": "Names `go mod tidy` from module root (bash block) and states root cause: 'reconciles go.sum against go.mod by fetching the exact module versions and checksums your build now requires'. Verification shown: `go build ./...`, `go vet ./...`, `go test -race ./...`. For conflicts it uses `go mod why -m`/`go mod graph` before hand-bumping. Explicitly warns against hand-editing go.sum and bumping the go directive."
204
+ },
205
+ "passed": true
206
+ },
207
+ {
208
+ "output": "Run this first:\n\n```bash\ngo mod tidy\n```\n\nThat resyncs `go.sum` (and `go.mod`) against the requirements already declared. The typical cause is exactly what you describe: you edited the version by hand, so `go.mod` requires the new version but `go.sum` still only has the old version's hashes \u2014 Go refuses to build rather than proceed without a verified hash.\n\nThen verify:\n\n```bash\ngo build ./...\ngo vet ./...\ngo test -race ./...\n```\n\nIf `go mod tidy` succeeds, you're done \u2014 the root cause was a stale checksum file, not a real version conflict.\n\nIf it *doesn't* resolve it, work through these:\n\n- **`go mod tidy` reports a conflict or tries to pull a different version than you set** \u2014 check why with `go mod why -m <module>` and `go mod graph`. A transitive dependency may require a version incompatible with your bump. Fix the actual version choice; don't paper over it.\n- **You bumped to a version that doesn't exist or isn't tagged** \u2014 `go mod tidy` will say so. Confirm the version string (`go list -m -versions <module>`).\n- **Private module fetch fails** \u2014 that's a `GOPRIVATE`/auth issue, not a `go.sum` issue; `go.sum` won't be written for a module Go can't fetch.\n\nOne caution while you're in there: don't add a `replace` directive to make a version conflict go away, and don't nudge the `go`/`toolchain` directive to dodge a compile error from the new version. If the bump genuinely needs a newer language version, that's a deliberate decision to state in the commit \u2014 not a silent side effect of fixing `go.sum`.\n\nReport it as: root cause (stale `go.sum` after a manual `go.mod` bump), fix (`go mod tidy`), and confirmation that build/vet/`test -race` all pass.",
209
+ "outputSha256": "46d19c47dabeff18277a343f5f39ea94658213439b68aff5f7937bc45f03493e",
210
+ "promptSha256": "c2c5c9df499cd7b1cf97b0043b920e2b1f07f48e7e636973e01a285741e0307f",
211
+ "deterministic": [],
212
+ "judge": {
213
+ "verdict": "pass",
214
+ "reason": "Names `go mod tidy` as the fix (run first, module root implied). States root cause: go.mod requires new version but go.sum only has old hashes. Verification: `go build ./...`, `go vet ./...`, `go test -race ./...`. For conflicts it uses `go mod why -m`/`go mod graph` before hand-bumping. Warns against editing go/toolchain directive and replace, not committing them."
215
+ },
216
+ "passed": true
217
+ },
218
+ {
219
+ "output": "Run `go mod tidy` from the module root. That re-resolves the requirements in `go.mod` and updates `go.sum` with the checksums for the new dependency version (and prunes stale entries for versions no longer needed).\n\n```bash\ngo mod tidy\ngo build ./...\ngo vet ./...\ngo test -race ./...\n```\n\nSome things to check once it finishes:\n\n- **`go mod tidy` changed more than you expected.** It not only fixes `go.sum` \u2014 it also adds any missing requirements your code imports and removes entries nothing imports anymore. Skim the `go.mod`/`go.sum` diff; if it pulled in a new *indirect* dependency or bumped something else, that is real information about your dependency graph, not noise to discard.\n- **It fails to resolve.** That means the version you bumped to conflicts with something else in the graph. Diagnose before hand-editing:\n ```bash\n go mod why -m <module>\n go mod graph | grep <module>\n ```\n Then decide whether the fix is a version choice in `go.mod` or a real `replace` directive (only if you intentionally want to override resolution, e.g. a patched fork \u2014 never as a way to paper over a conflict you do not understand).\n- **It pulled your `go` directive or toolchain forward.** If `go mod tidy` raised the `go` line because the new dependency requires a newer language version, that is a real constraint \u2014 confirm the project is okay moving its declared minimum toolchain before accepting it, rather than reverting the line and hoping it still builds.\n\nRoot cause here is almost always just that `go.sum` predates the `go.mod` bump \u2014 `go mod tidy` is the smallest correct fix. Only reach for anything more surgical if tidy reports an actual resolution error.",
220
+ "outputSha256": "5c057b8568478f511a16441e4ac9837940ab9315e05f6277f0c8cdc3850c3a62",
221
+ "promptSha256": "c2c5c9df499cd7b1cf97b0043b920e2b1f07f48e7e636973e01a285741e0307f",
222
+ "deterministic": [],
223
+ "judge": {
224
+ "verdict": "pass",
225
+ "reason": "Names `go mod tidy` from module root (criterion 1). States root cause: 'go.sum predates the go.mod bump' and tidy 'updates go.sum with the checksums for the new dependency version' (criterion 2). Verification shown: `go build ./...`, `go test -race ./...` (criterion 3). For conflicts it uses `go mod why -m`/`go mod graph` before hand-editing, and warns against reverting the go directive. No fail criteria met."
226
+ },
227
+ "passed": true
228
+ },
229
+ {
230
+ "output": "Run `go mod tidy`. That's the standard fix when `go.mod` was hand-edited (a dependency version bumped) without the corresponding `go.sum` update: `go mod tidy` reconciles `go.sum` against `go.mod`'s requirements, adding the checksums needed for the new version and pruning entries no longer needed.\n\nA couple of things worth checking while you're at it:\n\n- If `go mod tidy` still complains, the version you bumped to may not exist or may not be reachable \u2014 check the exact version string against the module's published tags (`go list -m -versions <module>`).\n- If `go mod tidy` reports that it's *changing* more than the one dependency you bumped, that's a signal other requirements had drifted too; let it settle, then review the `go.mod`/`go.sum` diff to confirm the only intentional change is the one you meant.\n- Don't reach for `GOFLAGS=-mod=mod` or `-insecure` to force the build through \u2014 that just suppresses the mismatch instead of resolving it.\n\nOnce tidy completes, verify with:\n\n```\ngo build ./...\ngo vet ./...\ngo test -race ./...\n```\n\nAll three should exit 0 before you call it done.",
231
+ "outputSha256": "fc1a1006befa7de05cc6d988e54c61729b6fdcd45de818002706f8d063a04a3a",
232
+ "promptSha256": "c2c5c9df499cd7b1cf97b0043b920e2b1f07f48e7e636973e01a285741e0307f",
233
+ "deterministic": [],
234
+ "judge": {
235
+ "verdict": "pass",
236
+ "reason": "Names `go mod tidy` as the fix (\"Run `go mod tidy`\"), states root cause (\"go.mod was hand-edited... without the corresponding go.sum update... reconciles go.sum against go.mod's requirements\"), and gives verification steps (`go build ./...`, `go test -race ./...`). No go-directive bump, no hand-editing hashes."
237
+ },
238
+ "passed": true
239
+ },
240
+ {
241
+ "output": "Run `go mod tidy`. That re-resolves the module graph and rewrites `go.sum` to match the new requirements in `go.mod`, adding the checksums for the bumped dependency and pruning entries for anything no longer needed.\n\nIf `go mod tidy` produces no change and the build still complains, the usual causes are:\n\n- `GOFLAGS=-mod=readonly` is set in your environment or CI config, which makes `go build` refuse to update `go.sum` at all. Run `go mod tidy` explicitly, or unset that flag for the fix.\n- `GONOSUMDB`/`GONOSUMCHECK`/`GOPRIVATE` (or `GOSUMDB=off`) is interfering with checksum verification for that module. Check `go env` for those.\n- A vendored tree (`vendor/` present) \u2014 then run `go mod tidy` and follow with `go mod vendor` (or `go mod tidy -e` first if a module is unreachable).\n\nBefore running it, check whether you actually meant to bump that dependency: `git diff go.mod` should show only the version you intended. `go mod tidy` will happily pull in whatever the new version's own requirements demand, so eyeball the resulting `go.mod` diff for surprise indirect upgrades.\n\nThen verify:\n\n```bash\ngo build ./...\ngo vet ./...\ngo test -race ./...\n```\n\nState the root cause in your commit/report \u2014 e.g. \"go.sum predated the X bump; ran go mod tidy\" \u2014 rather than just \"fixed go.sum.\" And don't reach for `-mod=mod` on every build or a `replace` directive to route around this; a stale `go.sum` after a deliberate bump is a normal tidy, not a version conflict needing an override.",
242
+ "outputSha256": "b177bada1e535326db2693335f933aecc17ec2db2b6386c3cc572d8b1e8b4ddc",
243
+ "promptSha256": "c2c5c9df499cd7b1cf97b0043b920e2b1f07f48e7e636973e01a285741e0307f",
244
+ "deterministic": [],
245
+ "judge": {
246
+ "verdict": "pass",
247
+ "reason": "Names `go mod tidy` as the fix (\"Run `go mod tidy`. That re-resolves the module graph and rewrites `go.sum` to match the new requirements in `go.mod`\"), states root cause (go.sum predated the bump), and gives verification (`go build ./...`, `go test -race ./...`). No go-directive bump, no hand-editing hashes."
248
+ },
249
+ "passed": true
250
+ },
251
+ {
252
+ "output": "Run:\n\n```bash\ngo mod tidy\n```\n\nThat reconciles `go.sum` against the new requirement(s) in `go.mod` \u2014 it re-derives the needed checksums, adds any missing transitive sums, and drops entries no longer referenced. Then re-verify:\n\n```bash\ngo build ./...\ngo vet ./...\ngo test -race ./...\n```\n\nA few things worth checking so this stays a one-line fix rather than a symptom-masking:\n\n- **Confirm the bump is intentional.** `go mod tidy` will happily make the build consistent with whatever version you typed into `go.mod`. Before running it, glance at the diff \u2014 if the new version pulled in a large set of new transitive modules or a surprise major-version jump, that's worth understanding, not just accepting.\n- **If `go mod tidy` itself fails or keeps changing `go.sum`,** you likely have a version conflict rather than a stale checksum. Inspect with `go mod why -m <module>` and `go mod graph` before hand-editing anything. Do not add a `replace` directive to route around the conflict unless it points at a real, intentional override (a local fork, a monorepo sibling) \u2014 and note that choice in the report if you do.\n- **Do not** bump the `go`/`toolchain` directive or edit `go.sum` by hand to make the error go away. Those change the module's declared requirements for every consumer instead of fixing the actual mismatch.\n\nOnce `go mod tidy` runs clean and the three verification commands pass, you're done. If they don't, the failure has moved from \"module metadata\" to something else (a real compile error, a toolchain mismatch) \u2014 classify that new error before touching `go.mod` again.",
253
+ "outputSha256": "688f84f0160bf32e5658315966ae52397e56cfc59cd60b8147a6be7a17f1ab74",
254
+ "promptSha256": "c2c5c9df499cd7b1cf97b0043b920e2b1f07f48e7e636973e01a285741e0307f",
255
+ "deterministic": [],
256
+ "judge": {
257
+ "verdict": "pass",
258
+ "reason": "Names `go mod tidy` as the concrete command (code block) and states root cause: 'reconciles go.sum against the new requirement(s) in go.mod... re-derives the needed checksums.' Verification shown: `go build ./...`, `go vet`, `go test -race`. For conflicts it names `go mod why -m` and `go mod graph` before hand-editing. Explicitly warns against bumping go/toolchain directive and hand-editing go.sum. No fail criteria hold."
259
+ },
260
+ "passed": true
261
+ },
262
+ {
263
+ "output": "Run `go mod tidy` \u2014 that's the direct fix when `go.sum` is simply stale against `go.mod`.\n\n`go.sum` is a record of the checksums for every module version the build actually needs, including transitive dependencies. When you bump a dependency version in `go.mod`, the new version's checksum (and possibly a few new transitive entries, or a removed old one) hasn't been added yet, so the build refuses to proceed.\n\nSteps:\n\n1. `go mod tidy` \u2014 this re-resolves the module graph, adds missing entries to `go.sum`, removes entries no longer needed, and may also prune `go.mod` of now-unused requirements.\n2. `go build ./...` and `go vet ./...` to confirm it's clean.\n3. `go test -race ./...` if there are tests affected by the bump.\n\nA couple of things worth checking if `go mod tidy` alone doesn't clear it:\n\n- If it fails with a version conflict rather than a checksum mismatch, that's a different problem \u2014 look at `go mod why -m <module>` and `go mod graph` before changing anything by hand. Don't reach for a `replace` directive to route around a conflict; only add one if it's a genuine, intentional override (a local sibling module, a patched fork), and note that in your report.\n- If `go mod tidy` wants to change the `go` directive or add a `toolchain` line, pause \u2014 that changes the minimum toolchain every consumer of the module needs. Only accept it if the dependency genuinely requires it; otherwise consider staying on the version the module already declares.\n- If you see a `checksum mismatch` for a specific module rather than a missing entry, that usually means a cached copy is corrupt or a module was re-tagged upstream \u2014 cle",
264
+ "outputSha256": "0d42c48dad8b9d4a21d8c34a3d739ed5859ee801d0916e990fe6804417ab6d31",
265
+ "promptSha256": "c2c5c9df499cd7b1cf97b0043b920e2b1f07f48e7e636973e01a285741e0307f",
266
+ "deterministic": [],
267
+ "judge": {
268
+ "verdict": "pass",
269
+ "reason": "Names `go mod tidy` as the fix (step 1) and explains root cause: go.sum records checksums for needed module versions; bumping a dep means the new version's checksum isn't added yet. Verification: `go build ./...`, `go vet ./...`, `go test -race ./...`. For version conflicts it says check `go mod why -m`/`go mod graph` before hand changes. No fail criteria met."
270
+ },
271
+ "passed": true
272
+ },
273
+ {
274
+ "output": "Run:\n\n```bash\ngo mod tidy\n```\n\nThat's the fix here. `go.sum` is just the checksum ledger for exactly what `go.mod` requires \u2014 when you bump a version in `go.mod` by hand, `go.sum` still has the old version's hashes and is missing the new one's, so the build refuses to proceed. `go mod tidy` resolves the full module graph, drops entries no longer needed, and writes the new checksums.\n\nA few cautions before you call it done:\n\n- **Check what tidy actually changed.** If it adds or removes a lot beyond the dependency you bumped, that usually means transitive requirements shifted; eyeball `git diff go.mod go.sum` rather than committing blind.\n- **If tidy fails or the version conflicts,** that's a different problem than \"out of date.\" Dig into it with `go mod why -m <module>` and `go mod graph` before bumping anything else by hand.\n- **Don't reach for a `replace` directive** just to make the mismatch go away \u2014 only use one for an intentional local override (a sibling module or a patched fork), and say so explicitly.\n- **Don't touch the `go` directive or bump other versions** to force it through; the mismatch is purely a checksum sync issue, not a toolchain issue.\n\nThen verify all three pass before you're done:\n\n```bash\ngo build ./...\ngo vet ./...\ngo test -race ./...\n```\n\nAnd if the project has a linter configured (`.golangci.yml` etc.), run `golangci-lint run ./...` too. Report it as: \"Fixed go.sum out of date after bumping <dependency> \u2014 ran `go mod tidy`; build/vet/test pass.\" State the root cause, not just \"build now passes.\"",
275
+ "outputSha256": "a2001086b5e448713225e4139dd568d9f58116b7a98dd9c61ebf7d32852885e8",
276
+ "promptSha256": "c2c5c9df499cd7b1cf97b0043b920e2b1f07f48e7e636973e01a285741e0307f",
277
+ "deterministic": [],
278
+ "judge": {
279
+ "verdict": "pass",
280
+ "reason": "Names `go mod tidy` as the fix (code block). Explains root cause: go.sum is checksum ledger for go.mod; bumping version leaves old hashes/missing new ones. Verification: `go build ./...`, `go vet ./...`, `go test -race ./...`. Warns against touching `go` directive/bumping versions and hand-editing; suggests `go mod why -m`/`go mod graph` for conflicts. No fail criteria met."
281
+ },
282
+ "passed": true
283
+ },
284
+ {
285
+ "output": "A stale `go.sum` after a `go.mod` change is expected \u2014 `go.sum` is a record of the exact module content hashes that `go.mod`'s requirements resolve to, and bumping a dependency changes what needs to be recorded. The fix is almost always:\n\n```bash\ngo mod tidy\n```\n\nThat reconciles `go.mod` and `go.sum` against the actual imports in your code: it adds the hashes for the newly required version, drops entries for versions nothing needs anymore, and adds/removes `go.mod` requirements so they match your real import graph.\n\nIf you want only to refresh `go.sum` without touching `go.mod`'s requirement set:\n\n```bash\ngo mod download\n```\n\nBut `go mod tidy` is the normal answer here, and it's what I'd run first.\n\nAfter running it, verify nothing else broke:\n\n```bash\ngo build ./...\ngo vet ./...\ngo test -race ./...\n```\n\nAll three should exit 0 before you consider this done.\n\nA couple of cautions:\n\n- If `go mod tidy` fails or keeps churning (`go.sum` still out of date on the next build), that usually means a genuine version conflict between transitive requirements, not just a stale hash. In that case diagnose before touching anything by hand:\n ```bash\n go mod why -m <module>\n go mod graph\n ```\n Understand which module is pulling in which version before changing a version manually.\n- Don't reach for a `replace` directive to route around this. A `replace` is for an intentional, documented override (a local monorepo sibling, a patched fork) \u2014 not a way to paper over version resolution.\n- Don't bump the `go`/`toolchain` directive just to make it compile. That changes the minimum toolchain for every consumer of the module; if the code genuinely needs a newer language feature, that's a separate, intentional decision.\n\nAlso worth a quick look at `git diff go.mod go.sum` after `go mod tidy`: confirm the only changes are the dependency you bumped plus its transitive additions, with no surprise downgrades or extra modules swept in.",
286
+ "outputSha256": "1afda5dc9b05248c57fceb76c9b8891c40b0a49d41b7f3320eda405b189e61ae",
287
+ "promptSha256": "c2c5c9df499cd7b1cf97b0043b920e2b1f07f48e7e636973e01a285741e0307f",
288
+ "deterministic": [],
289
+ "judge": {
290
+ "verdict": "pass",
291
+ "reason": "Names `go mod tidy` as the fix (code block), states root cause: go.sum records hashes go.mod's requirements resolve to and bumping changes what's recorded. Verification: `go build ./...`, `go vet ./...`, `go test -race ./...`. For conflicts, uses `go mod why -m`/`go mod graph` before manual version changes. Warns against bumping go/toolchain and replace; no hand-editing hashes."
292
+ },
293
+ "passed": true
294
+ }
295
+ ]
296
+ },
297
+ {
298
+ "id": "no-nolint-suppression",
299
+ "kind": "behavior",
300
+ "prompt": "golangci-lint is flagging an ignored error return in my Go code. How should I fix it?",
301
+ "strictness": "high",
302
+ "trials": 10,
303
+ "passes": 10,
304
+ "passRate": 1,
305
+ "passAtK": 1,
306
+ "grader": "judge",
307
+ "status": "ran",
308
+ "trialRecords": [
309
+ {
310
+ "output": "Fix the ignored error itself \u2014 don't silence the linter.\n\n**The rule:** lint findings like `errcheck` name a real thing (an error you're dropping). Address what it found; never add `//nolint` or a blank `_ =` discard whose only purpose is to make the checker stop.\n\n**Decide what the call's error actually needs, then do that:**\n\n- **Propagate it** if the caller should know:\n ```go\n if err := doThing(); err != nil {\n return fmt.Errorf(\"doing thing: %w\", err)\n }\n ```\n\n- **Handle it** if there's a real recovery (log, retry, fall back):\n ```go\n if err := doThing(); err != nil {\n log.Printf(\"doing thing failed, using default: %v\", err)\n // use the default\n }\n ```\n\n- **Explicitly acknowledge it** only when the error is genuinely, provably meaningless at that call site \u2014 and say why in the code so the next reader (and the linter) understands it's deliberate, not an oversight:\n ```go\n // Close's error is not actionable here; the write path already succeeded.\n _ = resp.Body.Close()\n ```\n `errcheck` generally accepts an explicit `_ =` assignment as \"I saw this and chose to ignore it\" \u2014 that's different from adding a `//nolint`, and it must be a decision you can defend, not a reflex.\n\n**What not to do:**\n- Don't add `//nolint:errcheck` to silence it \u2014 that also suppresses *future* real findings on that line.\n- Don't wrap the linter config to exclude the file or rule.\n- Don't retrofit a fake `panic(err)` just to use the variable.\n\n**Verify:** re-run `golangci-lint run ./...` \u2014 it should exit 0 because the error is now handled, not because the finding was suppressed. If the flag was `errcheck` and you chose propagation, confirm the wrapping still preserves `errors.Is`/`errors.As` behavior (use `%w`, not `%v`).",
311
+ "outputSha256": "834692a308a947ead006c0858a28d8df6aee3c7a6a57f6d6c1e7e72344acaa7e",
312
+ "promptSha256": "18f1fadd15fa89a73aa8a5a5748aeac4b4a068c9558ce1c74fc37ff667b37c3f",
313
+ "deterministic": [],
314
+ "judge": {
315
+ "verdict": "pass",
316
+ "reason": "Answer shows concrete wrap-and-return: `if err := doThing(); err != nil { return fmt.Errorf(\"doing thing: %w\", err) }` (criterion 1 & 2). Also gives a justified `_ = resp.Body.Close()` discard with a comment naming why it's safe. Explicitly warns against `//nolint:errcheck` and bare `_ =` as silencing (criterion 3). No fail criteria hold."
317
+ },
318
+ "passed": true
319
+ },
320
+ {
321
+ "output": "First, see what the error actually is. An ignored error return from `golangci-lint` (usually `errcheck`) means a call that returns an `error` has its result discarded. The fix depends on whether that error can meaningfully fail:\n\n**If the error can occur and matters** \u2014 handle it like any other error: return it up the stack, log it, or branch on it. This is the common case and the real fix.\n\n**If the error genuinely cannot fail in this call site**, make that explicit and justified rather than silently discarded. Options, in order of preference:\n\n1. Assign and check it, even if just to return it:\n ```go\n if _, err := w.Write(b); err != nil {\n return err\n }\n ```\n\n2. If it truly must be ignored, do so consciously and locally \u2014 never with a blanket `//nolint:errcheck`. If there's a real reason (e.g. a `Close()` on a read-only handle whose error is meaningless), a targeted `_ = f()` with a one-line comment explaining *why* it's safe to drop is acceptable. A bare `//nolint` is not \u2014 that's in the Red Flags for this skill: it silences the finding without addressing what it caught.\n\n**What not to do:**\n- Don't add `//nolint:errcheck` to make the linter stop complaining. That's suppressing a finding, not fixing it.\n- Don't add a blank `_ =` discard whose only purpose is to silence the checker with no reasoning behind it.\n- Don't disable the `errcheck` linter project-wide for one call site.\n\n**Verify:** re-run `golangci-lint run ./...` (plus `go build ./...`, `go vet ./...`, and `go test -race ./...` if you touched logic) and confirm all exit 0. The change should be local to the one call \u2014 no unrelated edits.\n\nIf you paste the exact linter output (the line, the flagged call, and the function signature), I can tell you which of the two cases you're in and the minimal edit.",
322
+ "outputSha256": "6fec312e4933a627631d644eec8d8a4120e8354255940a664703d7a7fe6664bb",
323
+ "promptSha256": "18f1fadd15fa89a73aa8a5a5748aeac4b4a068c9558ce1c74fc37ff667b37c3f",
324
+ "deterministic": [],
325
+ "judge": {
326
+ "verdict": "pass",
327
+ "reason": "Answer shows an `if _, err := w.Write(b); err != nil { return err }` check (criterion 1), and offers a concrete `_ = f()` discard with a one-line comment naming why it's safe (e.g. Close on read-only handle) (criterion 2). It explicitly rejects `//nolint:errcheck` and bare `_ =` as fixes (criterion 3), so no fail criteria hold."
328
+ },
329
+ "passed": true
330
+ },
331
+ {
332
+ "output": "Fix the ignored error \u2014 don't silence the linter.\n\nThe finding (usually `errcheck`) is telling you a function returned an `error` and you dropped it. The right fix depends on whether the error is actually actionable at that call site:\n\n**1. If the error matters, handle it:**\n\n```go\nif err := f.Close(); err != nil {\n return fmt.Errorf(\"closing output file: %w\", err)\n}\n```\n\n**2. If there's a genuinely correct, intentional discard, be explicit about it:**\n\n```go\n// Best-effort cleanup; the write result is what callers care about.\n_ = resp.Body.Close()\n```\n\nOnly do this when you can state *why* the error is safe to drop (e.g. a read-only `Close()` after a successful read, best-effort cleanup where a later step already reported the real failure). An explicit `_ =` with no reason is just as bad as `//nolint` \u2014 it's a suppression that happens to use legal syntax.\n\n**What NOT to do:**\n- Do not add `//nolint:errcheck`. That makes the checker quiet without addressing what it caught. Same for a blanket `//nolint` on the file or a config exclusion in `.golangci.yml` targeting the rule.\n- Do not discard just to get a green build. If you can't articulate why the error is droppable, it isn't.\n\n**Verify:**\n\n```bash\ngo build ./...\ngo vet ./...\ngolangci-lint run ./...\ngo test -race ./...\n```\n\nAll must exit 0. In your report, name the call site and the reason it was safe to handle or drop \u2014 \"added `_ =`\" is not a root-cause explanation.\n\nOne nuance: if the error is `errcheck` firing on deferred `Close()` in a function that already returns an error, the idiomatic fix is a named return plus a deferred check, or an explicit close before return \u2014 not a discard. Handle it the same way: decide whether the close error should surface, then make that decision visible in the code.",
333
+ "outputSha256": "69a9970374bef6a6c7b23f0e6fb9ef6077f6a439e36ffbe5df4fa887df3c1b34",
334
+ "promptSha256": "18f1fadd15fa89a73aa8a5a5748aeac4b4a068c9558ce1c74fc37ff667b37c3f",
335
+ "deterministic": [],
336
+ "judge": {
337
+ "verdict": "pass",
338
+ "reason": "Criterion 1: shows `if err := f.Close(); err != nil` branch. Criterion 2: concrete wrap-and-return `return fmt.Errorf(\"closing output file: %w\", err)` and a justified discard `_ = resp.Body.Close()` with comment naming why safe. Criterion 3: explicitly warns against `//nolint:errcheck` and bare `_ =` without reason. No fail criteria: no nolint recommendation, no log-only handling."
339
+ },
340
+ "passed": true
341
+ },
342
+ {
343
+ "output": "The right fix depends on what the ignored call actually is, but the guiding rule is: **make the error impossible to ignore by handling it, not by silencing the linter.** Never reach for `//nolint:errcheck` or a `_ =` discard as the fix.\n\nWalk through it like this:\n\n**1. Look at the exact line the linter names.**\nRun `golangci-lint run ./...` and read the finding. It will name the function/file/line. Identify which call's returned error is being dropped.\n\n**2. Ask what the error means at that call site, then pick the matching handling:**\n\n- **The error is real and the caller can act on it** \u2014 propagate it: `if err != nil { return fmt.Errorf(\"doing X: %w\", err) }`. This is the default when the function you're in already returns an error.\n- **The caller is the right place to decide** \u2014 return the error up rather than swallowing it, unless there's a genuine reason the operation is best-effort.\n- **The operation is genuinely best-effort** (e.g. a cleanup close, a metrics flush, a write to a log sink) \u2014 handle the failure in a way that's honest: log it, or explicitly discard it with a comment explaining *why* it's safe to ignore. Note that an explicit discard plus a reason is very different from a bare `_ =` whose only purpose is to quiet the checker.\n- **The error shouldn't exist in the first place** \u2014 e.g. it's an error you know is always nil for a specific input. Fix the API so it doesn't return an error, or handle it, rather than dropping it.\n\n**3. Common concrete cases:**\n- `defer f.Close()` on a file you've written to: if a write error matters, don't discard the close error \u2014 capture it and check it. For a read-only or already-flushed close, an explicit, commented discard can be legitimate.\n- `fmt.Fprintf`/`fmt.Println` into a buffer or `os.Stdout`: handle or intentionally document.\n- `json.Unmarshal`/type assertions where you already validated the input: if you truly know it can't fail, prefer restructuring so the error can't arise rather than ignoring it.\n\n**4. Fix the underlying issue the finding names**, which is exactly the discipline this skill enforces for `go vet`/`golangci-lint` findings.\n\n**5. Verify before calling it done:**\n```bash\ngo build ./...\ngo vet ./...\ngo test -race ./...\ngolangci-lint run ./...\n```\nAll must exit 0, and the change should touch only what the finding requires \u2014 no unrelated files, and no `//nolint` added to make the error go away.\n\nThe one-line summary: decide whether the error can/should be acted on at that site; if yes, propagate or handle it; if no, discard it *explicitly with a stated reason* \u2014 but never silence the linter instead of addressing what it found.",
344
+ "outputSha256": "2148127f34e478d86aeaa31df024d5db10e2e044457bf6b4a0a882bed55006c9",
345
+ "promptSha256": "18f1fadd15fa89a73aa8a5a5748aeac4b4a068c9558ce1c74fc37ff667b37c3f",
346
+ "deterministic": [],
347
+ "judge": {
348
+ "verdict": "pass",
349
+ "reason": "Answer explicitly checks the error with `if err != nil { return fmt.Errorf(\"doing X: %w\", err) }` (criterion 1 & 2, concrete wrap-and-return). It also shows the discard shape: 'explicitly discard it with a comment explaining why it's safe.' It warns against `//nolint:errcheck` and bare `_ =` as the fix (criterion 3), and never relies on logging alone without justification."
350
+ },
351
+ "passed": true
352
+ },
353
+ {
354
+ "output": "Fix the ignored error, don't suppress the finding. The specific fix depends on what the error *means* at that call site \u2014 that's the root cause question, and the linter is just the messenger.\n\n**The three real fixes, in order of preference:**\n\n1. **Handle it.** Check the error and do something meaningful (return it, log with context, wrap and propagate). This is correct whenever the caller can actually recover from or report the failure.\n\n```go\nif err := f.Close(); err != nil {\n return fmt.Errorf(\"close config: %w\", err)\n}\n```\n\n2. **Explicitly ignore it, and document why it's safe.** Some returns genuinely cannot fail meaningfully (`(*bytes.Buffer).Write`, `fmt.Fprintf` to a buffer, `json.Encoder.Encode` to a bytes.Buffer, `os.Remove` on a temp file best-effort cleanup). Assign to `_` *only* when you can state why the error is unrecoverable or irrelevant \u2014 and ideally leave a one-line comment saying so.\n\n```go\n_ = tmp.Close() // best-effort cleanup; nothing to recover from here\n```\n\n3. **Restructure so there's no error to ignore** \u2014 e.g. use a call that doesn't return one, or collect cleanup errors in a `defer` and surface them (the `errors.Join` idiom):\n\n```go\ndefer func() { err = errors.Join(err, f.Close()) }()\n```\n\n**What not to do:** adding `//nolint:errcheck` (or `//nolint` with no code) whose only purpose is to silence the finding. That defeats the point of the check and hides exactly the failure mode it was written to catch. If you're tempted to add it, that's the signal you haven't decided what the error means yet \u2014 decide first.\n\n**One caveat that makes this unambiguous:** golangci-lint's `errcheck` should be flagging an *unchecked* return. If you already assigned it to `_` on purpose and the linter still complains, that's a config mismatch (`check-blank` / exclusion rules), not a code problem \u2014 fix the config, don't add more suppression. But if the flagged line has a naked call whose error is simply dropped, the fix is one of the three above.\n\n**Then verify:** `golangci-lint run ./...`, `go build ./...`, and `go test -race ./...` all exit 0, and the diff touches only the call site in question. If you can't write one sentence explaining *why* the error is safe to ignore at that specific line, you haven't actually fixed it yet.",
355
+ "outputSha256": "c4f5fa437da32d23ece84730b2b6fb20b082b5918c963953723f409124f02fd3",
356
+ "promptSha256": "18f1fadd15fa89a73aa8a5a5748aeac4b4a068c9558ce1c74fc37ff667b37c3f",
357
+ "deterministic": [],
358
+ "judge": {
359
+ "verdict": "pass",
360
+ "reason": "Answer shows concrete fixes: wrap-and-return snippet `return fmt.Errorf(\"close config: %w\", err)` inside `if err := f.Close(); err != nil` (criteria 1,2), and a `_ = tmp.Close() // best-effort cleanup...` discard with justifying comment (criterion 2). It explicitly warns against `//nolint:errcheck` as suppression (criterion 3), not recommending it. No fail criteria hold."
361
+ },
362
+ "passed": true
363
+ },
364
+ {
365
+ "output": "The fix depends on what the ignored error actually means at that call site \u2014 `golangci-lint` (via `errcheck`) is telling you a returned error was discarded, so the question to answer first is: *should this error be handled, and how?* Do **not** reach for `//nolint:errcheck` or a blank `_ =` discard; that silences the finding without addressing what it caught.\n\nClassify the call site:\n\n**1. The error is genuinely actionable** (file write, network call, DB exec, `json.Marshal`, `template.Execute` to a real writer). Handle it:\n\n```go\nif err := f.Close(); err != nil {\n return fmt.Errorf(\"close %s: %w\", path, err)\n}\n```\n\nWrap with context and `%w` so the caller can inspect it. This is the common case.\n\n**2. The error can only happen in a situation you can't recover from, but you still shouldn't silently drop it.** For a `Close()` on a read-only file, or a deferred cleanup where the primary operation already failed, the idiomatic move is to still check it and either join it into the returned error (`errors.Join`) or log it explicitly:\n\n```go\ndefer func() {\n if cerr := rc.Close(); cerr != nil {\n err = errors.Join(err, cerr)\n }\n}()\n```\n\n**3. The error truly is irrelevant** (rare \u2014 e.g. `fmt.Fprintln(os.Stderr, ...)` where you genuinely don't care, or `defer tx.Rollback()` after a successful `Commit`, where rollback returning \"already committed\" is expected). Only here is an explicit `_ =` acceptable, and it should carry a one-line comment saying why:\n\n```go\n// Rollback after a successful Commit is a no-op; the error is expected.\ndefer func() { _ = tx.Rollback() }()\n```\n\nEven then, prefer checking it when cheap.\n\n**4. It's a `defer`-inside-a-loop or errcheck-on-`defer` config issue, not a real ignored error.** Re-read the exact lint message and the line it points at before changing anything \u2014 sometimes the finding is about a `defer`'d call whose error genuinely cannot be surfaced at that point, which is case 3, not case 1.\n\nWhat you should **not** do:\n\n- Add `//nolint:errcheck` to make it go away \u2014 the finding exists because ignoring errors is a real bug source, and suppressing it defeats the check.\n- Add `_ = someCall()` purely to satisfy the linter when the error is actionable \u2014 that's the same bug with extra steps.\n- Assert the error is \"always nil\" without verifying it against the actual implementation or docs.\n\nAfter the fix, re-run `golangci-lint run ./...` (and `go build ./...`, `go vet ./...`, `go test -race ./...`) and confirm it exits 0. State in your report which of the four cases this was and why the chosen handling is correct \u2014 \"fixed the lint error\" without naming what the error meant usually means it wasn't actually fixed.",
366
+ "outputSha256": "4c383ac44e55c4f603464db190e0b394607aa6f6cd14143a2c9918524a558e7c",
367
+ "promptSha256": "18f1fadd15fa89a73aa8a5a5748aeac4b4a068c9558ce1c74fc37ff667b37c3f",
368
+ "deterministic": [],
369
+ "judge": {
370
+ "verdict": "pass",
371
+ "reason": "Case 1 shows concrete wrap-and-return: `if err := f.Close(); err != nil { return fmt.Errorf(\"close %s: %w\", path, err) }`. Case 3 shows explicit `_ = tx.Rollback()` with a comment naming why it's safe. Explicitly warns against //nolint and bare `_ =` as the fix. No logging-only handling."
372
+ },
373
+ "passed": true
374
+ },
375
+ {
376
+ "output": "Handle the error \u2014 the right fix depends on what the function actually promises and what the caller can do about the failure.\n\n**1. Read the exact finding first.**\n\n```bash\ngolangci-lint run ./...\n```\n\nIt will name the checker (`errcheck`), the file:line, and the expression (e.g. `json.Unmarshal(...)` return value not checked). The checker is telling you a returned `error` was silently discarded. The question is what *should* happen to it.\n\n**2. Pick the fix that matches the semantics.**\n\n- **The error is real and actionable** \u2014 propagate it:\n ```go\n if err := doThing(); err != nil {\n return fmt.Errorf(\"do thing: %w\", err)\n }\n ```\n Wrap with `%w` so callers can match it, per the project's error-wrapping convention.\n\n- **The error is real but this is a cleanup/close where you can't fail the caller's flow** (e.g. a deferred `Close`) \u2014 surface it rather than discard it:\n ```go\n defer func() {\n if err := f.Close(); err != nil {\n logger.Printf(\"close %s: %v\", name, err)\n }\n }()\n ```\n Logging is honest; silently dropping it is not.\n\n- **The error is genuinely impossible/irrelevant at this call site** and the project already has a sanctioned idiom for that \u2014 the *explicit* discard, with a comment saying why:\n ```go\n // Content-Type is optional; a missing header is fine here.\n _ = req.ParseForm()\n ```\n Note this is different from `_ =` used purely to mute the linter.\n\n**3. Never do the thing the finding is actually about.**\n\n```go\n// WRONG: mutes the checker, discards a real error\n//nolint:errcheck\ndoThing()\n```\n\nAdding `//nolint:errcheck` (or `_ = doThing()`) purely to make the checker stop is the one fix this skill forbids \u2014 it hides the ignored error the checker caught instead of deciding what should happen to it. If a specific finding is a true false positive, that's a per-line `//nolint:errcheck // <reason>` with a real reason, but reach for it only after you've confirmed there's no correct handling available, and say so in your report.\n\n**4. Verify.**\n\n```bash\ngo build ./...\ngo vet ./...\ngo test -race ./...\ngolangci-lint run ./...\n```\n\nAll must exit 0. Then report the root cause in one sentence, e.g. *\"`errcheck` flagged `os.Chdir` in `Run`; the error was actionable, so it's now wrapped and returned.\"` \u2014 not just \"linter passes now.\"",
377
+ "outputSha256": "c09c794accc54590472b89f75cca526cbd6edd624c984f2ca6e6bfcd096bdb5e",
378
+ "promptSha256": "18f1fadd15fa89a73aa8a5a5748aeac4b4a068c9558ce1c74fc37ff667b37c3f",
379
+ "deterministic": [],
380
+ "judge": {
381
+ "verdict": "pass",
382
+ "reason": "Answer shows explicit `if err := doThing(); err != nil { return fmt.Errorf(\"do thing: %w\", err) }` (criterion 1 & 2), and an explicit discard `_ = req.ParseForm()` with comment naming why it's safe. It warns against `//nolint:errcheck` as the fix (criterion 3). Logging case is for deferred Close, not the sole handling of the flagged error."
383
+ },
384
+ "passed": true
385
+ },
386
+ {
387
+ "output": "Fix the ignored error, not the lint rule. `golangci-lint` (via `errcheck`) is telling you a returned `error` was discarded, which means a real failure path is being silently swallowed. The correct change is to actually handle it.\n\n**Preferred, in order:**\n\n1. **Handle it directly** \u2014 propagate or act on it:\n ```go\n if err := f.Close(); err != nil {\n return fmt.Errorf(\"close %s: %w\", name, err)\n }\n ```\n This is the goal whenever the error is meaningful (I/O, network, writes, parsing).\n\n2. **Propagate to the caller** if this function can't do anything useful with it yet \u2014 return it, wrapped with `%w` so the chain is preserved.\n\n3. **Log it, if the operation is genuinely best-effort** \u2014 e.g. a deferred close on a read path:\n ```go\n defer func() {\n if err := f.Close(); err != nil {\n log.Printf(\"close %s: %v\", name, err)\n }\n }()\n ```\n Logging is fine when the caller truly cannot act on the error; it is not a dumping ground for errors you don't feel like handling.\n\n4. **Assign to `_` only when the error is documented as impossible to matter** \u2014 and even then, pair it with a comment explaining why (`// error is documented as always nil for this reader`). A bare `_ =` with no comment is the thing to avoid.\n\n**Do not do this:**\n\n- Add `//nolint:errcheck` to silence the finding. That hides the exact thing the checker exists to surface.\n- Blank-assign (`_ = doThing()`) just to reach a green lint. If you're reaching for this because the fix is \"annoying,\" that's the signal the error path was actually being ignored.\n\n**One caveat worth checking first:** `errcheck` sometimes fires on deferred calls that are easy to overlook:\n```go\ndefer f.Close() // flagged: Close's error is dropped\n```\nThose are legitimate hits \u2014 the assignment to `defer` discards the return. Handle them with the wrapped/deferred form above.\n\nAlso, before shifting blame to the linter, confirm the toolchain state is sane:\n```bash\ngo mod tidy\ngo build ./...\ngo vet ./...\n```\nA stale `go.mod`/`go.sum` can produce findings that look lint-related but are actually version skew. Once the tree is clean, re-run `golangci-lint run ./...` and fix what remains at the source.\n\nIf you paste the exact finding (file, line, and the code), I can tell you which of the four options is right for that specific call.",
388
+ "outputSha256": "5bfb13a3c53ca41119b8e12423a95c6ca8968aa5b99dfa9293d2126460f6a509",
389
+ "promptSha256": "18f1fadd15fa89a73aa8a5a5748aeac4b4a068c9558ce1c74fc37ff667b37c3f",
390
+ "deterministic": [],
391
+ "judge": {
392
+ "verdict": "pass",
393
+ "reason": "Criterion 1: shows `if err := f.Close(); err != nil` branch. Criterion 2: concrete wrap-and-return `return fmt.Errorf(\"close %s: %w\", name, err)` and a `_ =` discard paired with a comment naming why it's safe. Criterion 3: explicitly warns against `//nolint:errcheck` and bare `_ =`. No fail criteria: logging is offered only for best-effort cases with justification, not as the sole fix."
394
+ },
395
+ "passed": true
396
+ },
397
+ {
398
+ "output": "Look at the actual error being ignored before touching the linter config or adding a suppression.\n\n**The fix is almost always one of these:**\n\n1. **The error genuinely needs handling** \u2014 wrap and return it, or if you're in a spot where you can't return it, at least log it with context:\n ```go\n if err := f.Close(); err != nil {\n return fmt.Errorf(\"closing file: %w\", err)\n }\n ```\n\n2. **The error is genuinely safe to ignore, and that's knowable from the context** \u2014 then say so explicitly rather than silently:\n ```go\n // Best-effort: connection is already being torn down; a failed\n // write-back here has nothing meaningfully to do about it.\n _ = conn.WriteFinalMessage()\n ```\n A bare `_ =` with no comment is what the linter (correctly) suspects; an _explained_ discard is the fix, not a suppression.\n\n3. **The call is in a deferred cleanup** where returning is impossible:\n ```go\n defer func() {\n if err := tx.Rollback(); err != nil && !errors.Is(err, sql.ErrTxDone) {\n log.Printf(\"rollback failed: %v\", err)\n }\n }()\n ```\n\n**What not to do:**\n\n- Do **not** add `//nolint:errcheck`. That silences the checker without addressing what it caught \u2014 if the error truly is ignorable, the explanation belongs in a plain comment (option 2), not a linter directive.\n- Do **not** reach for `//nolint` \"just this once to unblock the build.\" The finding is the linter doing its job.\n\n**If you're not sure whether the error matters**, that uncertainty is the answer: check what the function's error actually means (read its doc comment / source). If it can only fail for reasons that are already handled or impossible at this call site, option 2 with a comment is correct. If it can fail for a real reason, option 1 is correct. `golangci-lint` flagging it is a prompt to make that decision consciously, not to make the warning disappear.\n\nAfter the fix, re-run `golangci-lint run ./...` plus `go build ./...` and `go vet ./...` \u2014 the finding should be gone because the underlying issue is resolved, not because it's hidden.",
399
+ "outputSha256": "ca994ff38385d6ea64c650426c77a0644e99fde5d253c7dfa5b3db8e794dc01a",
400
+ "promptSha256": "18f1fadd15fa89a73aa8a5a5748aeac4b4a068c9558ce1c74fc37ff667b37c3f",
401
+ "deterministic": [],
402
+ "judge": {
403
+ "verdict": "pass",
404
+ "reason": "Option 1 shows concrete wrap-and-return: `if err := f.Close(); err != nil { return fmt.Errorf(\"closing file: %w\", err) }` (crit 1,2). Option 2 shows `_ = conn.WriteFinalMessage()` with a comment explaining why the discard is safe (crit 2). Explicitly warns against `//nolint:errcheck` (crit 3). Option 3's log-only is a deferred cleanup with justification, not the sole fix."
405
+ },
406
+ "passed": true
407
+ },
408
+ {
409
+ "output": "Handle the error at the call site instead of silencing the linter. Concretely:\n\n1. **Reproduce and read the exact finding.** Run the project's configured linter (e.g. `golangci-lint run ./...`) and note the rule (`errcheck`, `staticcheck`, etc.), the file/line, and which return value is ignored.\n\n2. **Fix the underlying issue the finding names.** The linter is telling you a call returned an error you discarded. Pick the right response for that call:\n - If the error is meaningful to the caller, return/wrap and propagate it (`if err != nil { return fmt.Errorf(\"doing X: %w\", err) }`).\n - If it's genuinely expected and recoverable, handle it explicitly (retry, fallback, log at the right level with context).\n - If the operation truly cannot fail in a way that matters here, only then assign it to a named discard \u2014 `_ = f()` \u2014 and make sure that is an intentional, documented decision rather than a reflex.\n\n3. **Never add `//nolint:errcheck`** (or a bare `_ =`) whose only purpose is to make the checker stop complaining. That hides the error the rule caught. This is the single most common mistake with this finding, and it is a red-flag rationalization, not a fix.\n\n4. **Verify.** `go build ./...`, `go vet ./...`, and your linter must all exit 0. `go test -race ./...` should pass too if the surrounding code touches shared state.\n\n5. **Report the root cause in one sentence** \u2014 e.g. \"Ignored the `Close()` error on the response body; now deferring a closure that logs failures instead of discarding them.\" Not just \"linter is green now.\"\n\nIf you're unsure which handling is correct because the call site lacks context (e.g. inside a deferred function or a goroutine with no error path out), that itself is worth resolving before choosing between propagation and an explicit discard \u2014 don't guess your way to a suppressed finding.",
410
+ "outputSha256": "f0ddbd29a85b8577539e7374877731e8f2dea6a70e86f98d45a77b75edcbd7f3",
411
+ "promptSha256": "18f1fadd15fa89a73aa8a5a5748aeac4b4a068c9558ce1c74fc37ff667b37c3f",
412
+ "deterministic": [],
413
+ "judge": {
414
+ "verdict": "pass",
415
+ "reason": "Criterion 1: shows `if err != nil { return fmt.Errorf(\"doing X: %w\", err) }`. Criterion 2: concrete wrap-and-return snippet plus `_ = f()` discard with documented-decision note. Criterion 3: explicitly rejects `//nolint:errcheck` and bare `_ =` as fixes. No fail criteria: no nolint recommendation, no log-only handling."
416
+ },
417
+ "passed": true
418
+ }
419
+ ]
420
+ }
421
+ ],
422
+ "verdict": "pass",
423
+ "scope": "bundled",
424
+ "skillDigest": "0bda83c5943fd515ce613009c345e8edc57a561913802a164e2b8fd2046e4b02",
425
+ "catalogDigest": "4f4016d410837e394a27e5b247e38ef2f57a1ee0baba4436ce7d3d71e223333d",
426
+ "judgePromptVersion": "2026-09-25.1",
427
+ "runner": "deepseek",
428
+ "model": "deepseek-chat",
429
+ "runnerPromptVersion": "2026-09-25.1",
430
+ "recordedAt": "2026-09-25T05:34:39.023Z",
431
+ "judge": "deepseek",
432
+ "judgeModel": "deepseek-chat"
433
+ },
434
+ {
435
+ "schemaVersion": "1.0.0",
436
+ "skillId": "go/go-code-review",
437
+ "strictness": "high",
438
+ "trials": 10,
439
+ "triggerAccuracy": {
440
+ "truePositive": 6,
441
+ "falsePositive": 0,
442
+ "positives": 6,
443
+ "negatives": 6
444
+ },
445
+ "evidence": "authored",
446
+ "scenarios": [
447
+ {
448
+ "id": "trigger-positive-1",
449
+ "kind": "trigger-positive",
450
+ "prompt": "Review this Go pull request for goroutine leaks",
451
+ "strictness": "high",
452
+ "trials": 1,
453
+ "passes": 1,
454
+ "passRate": 1,
455
+ "passAtK": 1,
456
+ "grader": "trigger-rank-fork-family",
457
+ "status": "ran",
458
+ "deterministic": true
459
+ },
460
+ {
461
+ "id": "trigger-positive-2",
462
+ "kind": "trigger-positive",
463
+ "prompt": "Check this Go diff for data races and ignored errors",
464
+ "strictness": "high",
465
+ "trials": 1,
466
+ "passes": 1,
467
+ "passRate": 1,
468
+ "passAtK": 1,
469
+ "grader": "trigger-rank-fork-family",
470
+ "status": "ran",
471
+ "deterministic": true
472
+ },
473
+ {
474
+ "id": "trigger-positive-3",
475
+ "kind": "trigger-positive",
476
+ "prompt": "Does this Go code store context.Context on a struct incorrectly?",
477
+ "strictness": "high",
478
+ "trials": 1,
479
+ "passes": 1,
480
+ "passRate": 1,
481
+ "passAtK": 1,
482
+ "grader": "trigger-rank-fork-family",
483
+ "status": "ran",
484
+ "deterministic": true
485
+ },
486
+ {
487
+ "id": "trigger-positive-4",
488
+ "kind": "trigger-positive",
489
+ "prompt": "Review this Go concurrency change for nil map writes",
490
+ "strictness": "high",
491
+ "trials": 1,
492
+ "passes": 1,
493
+ "passRate": 1,
494
+ "passAtK": 1,
495
+ "grader": "trigger-rank-fork-family",
496
+ "status": "ran",
497
+ "deterministic": true
498
+ },
499
+ {
500
+ "id": "trigger-positive-5",
501
+ "kind": "trigger-positive",
502
+ "prompt": "Check for interface pollution in this Go package",
503
+ "strictness": "high",
504
+ "trials": 1,
505
+ "passes": 1,
506
+ "passRate": 1,
507
+ "passAtK": 1,
508
+ "grader": "trigger-rank-fork-family",
509
+ "status": "ran",
510
+ "deterministic": true
511
+ },
512
+ {
513
+ "id": "trigger-positive-6",
514
+ "kind": "trigger-positive",
515
+ "prompt": "Review this Go handler for defer inside a loop",
516
+ "strictness": "high",
517
+ "trials": 1,
518
+ "passes": 1,
519
+ "passRate": 1,
520
+ "passAtK": 1,
521
+ "grader": "trigger-rank-fork-family",
522
+ "status": "ran",
523
+ "deterministic": true
524
+ },
525
+ {
526
+ "id": "trigger-negative-1",
527
+ "kind": "trigger-negative",
528
+ "prompt": "Review this Go code and also fix the bugs you find",
529
+ "strictness": "high",
530
+ "trials": 1,
531
+ "passes": 1,
532
+ "passRate": 1,
533
+ "passAtK": 1,
534
+ "grader": "trigger-rank-fork-family",
535
+ "status": "ran",
536
+ "deterministic": true
537
+ },
538
+ {
539
+ "id": "trigger-negative-2",
540
+ "kind": "trigger-negative",
541
+ "prompt": "Review this Python code for SQL injection",
542
+ "strictness": "high",
543
+ "trials": 1,
544
+ "passes": 1,
545
+ "passRate": 1,
546
+ "passAtK": 1,
547
+ "grader": "trigger-rank-fork-family",
548
+ "status": "ran",
549
+ "deterministic": true
550
+ },
551
+ {
552
+ "id": "trigger-negative-3",
553
+ "kind": "trigger-negative",
554
+ "prompt": "Review this TypeScript diff for missing null checks",
555
+ "strictness": "high",
556
+ "trials": 1,
557
+ "passes": 1,
558
+ "passRate": 1,
559
+ "passAtK": 1,
560
+ "grader": "trigger-rank-fork-family",
561
+ "status": "ran",
562
+ "deterministic": true
563
+ },
564
+ {
565
+ "id": "trigger-negative-4",
566
+ "kind": "trigger-negative",
567
+ "prompt": "Implement a bounded worker pool in this Go package using errgroup",
568
+ "strictness": "high",
569
+ "trials": 1,
570
+ "passes": 1,
571
+ "passRate": 1,
572
+ "passAtK": 1,
573
+ "grader": "trigger-rank-fork-family",
574
+ "status": "ran",
575
+ "deterministic": true
576
+ },
577
+ {
578
+ "id": "trigger-negative-5",
579
+ "kind": "trigger-negative",
580
+ "prompt": "Run a general security review on this codebase",
581
+ "strictness": "high",
582
+ "trials": 1,
583
+ "passes": 1,
584
+ "passRate": 1,
585
+ "passAtK": 1,
586
+ "grader": "trigger-rank-fork-family",
587
+ "status": "ran",
588
+ "deterministic": true
589
+ },
590
+ {
591
+ "id": "trigger-negative-6",
592
+ "kind": "trigger-negative",
593
+ "prompt": "Review this Go diff for naming conventions and formatting only",
594
+ "strictness": "high",
595
+ "trials": 1,
596
+ "passes": 1,
597
+ "passRate": 1,
598
+ "passAtK": 1,
599
+ "grader": "trigger-rank-fork-family",
600
+ "status": "ran",
601
+ "deterministic": true
602
+ },
603
+ {
604
+ "id": "read-only-review",
605
+ "kind": "behavior",
606
+ "prompt": "Review this Go diff: a struct field `ctx context.Context` was added and populated in the constructor. What do you find?",
607
+ "strictness": "high",
608
+ "trials": 10,
609
+ "passes": 10,
610
+ "passRate": 1,
611
+ "passAtK": 1,
612
+ "grader": "judge",
613
+ "status": "ran",
614
+ "trialRecords": [
615
+ {
616
+ "output": "**Finding: context stored on a struct field \u2014 `context.Context` as a struct field**\n\nFile:line \u2014 wherever the struct is declared (e.g. `internal/worker/pool.go:18`, the `ctx context.Context` field), and the constructor line where it is populated (e.g. `pool.go:31`, `ctx: ctx`).\n\nPattern: `context.Context` is stored as a struct field and assigned once in the constructor, then presumably read later from methods on that struct.\n\nWhy it matters (context misuse \u2014 Step 2, \"Context\"):\n- The context on the struct is fixed at construction time. Any cancellation, deadline, or value that arrives *after* the constructor runs \u2014 a request-scoped `r.Context()`, a shorter deadline set by a caller \u2014 will never be seen by methods that use the stored `p.ctx`. The context is decoupled from the operation it is supposed to govern.\n- It breaks propagation: cancellation and deadlines are meant to flow *down the call tree per call*, and storing one context freezes a single lifetime for every method on the value, which is exactly the anti-pattern the rule flags.\n- It also makes the context a value captured into whatever lifetime the struct has \u2014 if the struct outlives the originating request, the stored context keeps it alive and any values/tracing attached to it, and a later method that selects on `p.ctx.Done()` may hang forever (no one cancels the stored context) or fire at the wrong time.\n\nFix direction (not applied \u2014 read-only review):\n- Do not store the context. Remove the `ctx` field and take `ctx context.Context` as the **first parameter** of each method or function that needs it (`func (p *Pool) Run(ctx context.Context) error`), passing the caller's context through rather than a captured one.\n- The constructor should not accept and retain a context; construction is not request-scoped.\n- If the concern is \"methods need a context but nothing passes one,\" the fix is to thread the context through the call sites, not to hoist one onto the struct.\n\nVerify before finalizing:\n- Check every method that now reads `p.ctx` and confirm what context each caller should instead be passing \u2014 the review should confirm call sites exist that can supply a per-call context. If some entry point truly has none, note that as a separate finding (likely a `context.Background()` at a boundary) rather than as justification for the field.\n- Confirm whether this field is newly introduced by the diff or pre-existing in a merely-touched file; report it accordingly.\n\nNo source files were modified by this review.",
617
+ "outputSha256": "7f9f48c6c840c71253fa260fb07375a9287837fe82f1f1c56cbb63b2497ca361",
618
+ "promptSha256": "191412bc772d9ae6ba34287c6b44044bcf167e45ec42f96318bd432524857cf5",
619
+ "deterministic": [],
620
+ "judge": {
621
+ "verdict": "pass",
622
+ "reason": "Names the `ctx context.Context` struct field populated in the constructor (pool.go:18/31) as the problem (crit 1). Fix direction: remove field, take `ctx context.Context` as first param, thread caller's context (crit 2). Explicitly 'not applied \u2014 read-only review' and 'No source files were modified' (crit 3). No edit claimed."
623
+ },
624
+ "passed": true
625
+ },
626
+ {
627
+ "output": "Finding \u2014 struct field storing a `context.Context`, populated at construction.\n\nThis is the context-misuse case this review is meant to catch, and it is introduced by this diff (the field and its constructor assignment are both new, not pre-existing).\n\nWhy it matters:\n\n- `context.Context` carries request-scoped deadline, cancellation, and trace values. A context is only meaningful for the duration of one call chain, not for the lifetime of a long-lived struct. Storing it on a struct means the struct outlives the context's validity window, so every later method call on that struct runs under a context that may already be cancelled or past its deadline.\n- A struct with a stored `ctx` field usually ends up using a context that is unrelated to the operation actually being performed at call time. If the struct is constructed once and reused across many requests, later callers lose their own cancellation/deadline, and the original request's cancellation silently leaks into unrelated work.\n- If the field is populated with `context.Background()` or `context.TODO()` at construction (rather than the constructing caller's own context), cancellation and deadlines cannot propagate at all into that struct's downstream calls \u2014 a handler's `r.Context()` never reaches the work.\n- It hides the context dependency from method signatures, so callers and reviewers cannot tell that a method's cancellation behavior is controlled somewhere else entirely.\n\nFix direction (report only, not applied):\n\n- Remove the `ctx context.Context` field from the struct.\n- Passing the context through the method is the convention here, so take it as the leading parameter of each method that needs it: `func (s *Svc) Do(ctx context.Context, ...) error`. Do not store it on the struct or any other struct.\n- In HTTP handlers and request-scoped paths, pass `r.Context()` (or the caller-provided context) down rather than a fresh `context.Background()`/`context.TODO()`.\n- The only legitimate stored-context exceptions are rare and deliberate (for example some long-lived background workers that own their own lifetime context); if the constructor is doing that, that ownership and lifetime must be stated explicitly \u2014 otherwise treat this as a defect.\n\nVerification note: I cannot confirm from this snippet alone what value is assigned to the field in the constructor, or whether callers already pass their own context through. If the constructor assignment is `context.Background()`/`context.TODO()`, that is a second, compounding finding (background context used where a caller's context exists). Both should be checked at the constructor's `file:line` and in the method bodies that read `s.ctx`.\n\nNo source was modified by this review \u2014 the fix direction above is for the author to apply.",
628
+ "outputSha256": "c014a1edabe4952c1de9abd517acd07e7183ac285de34b6396e220b1ff1fb024",
629
+ "promptSha256": "191412bc772d9ae6ba34287c6b44044bcf167e45ec42f96318bd432524857cf5",
630
+ "deterministic": [],
631
+ "judge": {
632
+ "verdict": "pass",
633
+ "reason": "Names the new `ctx context.Context` struct field populated in the constructor as the problem (finding header + 'field and its constructor assignment are both new'). Fix direction: remove the field, take ctx as leading method param `func (s *Svc) Do(ctx context.Context, ...)`, pass `r.Context()` down. Explicitly read-only: 'report only, not applied' and 'No source was modified by this review.' No edit claimed."
634
+ },
635
+ "passed": true
636
+ },
637
+ {
638
+ "output": "A stored `context.Context` struct field is a finding under the Context category of the review \u2014 flag it, don't fix it.\n\n**Finding**\n\n```\n<file:line> \u2014 struct field `ctx context.Context` added and populated in the\nconstructor. Pattern: context stored on a struct. Risk: the context's\nlifetime no longer matches the call's lifetime \u2014 the struct outlives the\ncaller's cancellation, so a deadline/cancel that should bound one operation\ninstead persists for the struct's whole life (or leaks the cancel func's\nresources). Fix direction: remove the field; pass `ctx` as the first\nparameter to the methods that need it, or accept a per-call context.\n```\n\n**Why it matters**\n\nGo's rule is that a context belongs to a *call*, not an object. Putting it on the struct means every method implicitly uses whatever context was captured at construction. That breaks three things at once:\n\n- **Cancellation**: a `cancel()` from the constructing call no longer bounds the work the struct does later.\n- **Deadlines**: a deadline set for one request silently applies to unrelated later operations (or never expires).\n- **Tracing/values**: whatever `context.Value` payloads were live at construction bleed into calls made under a different request.\n\nIt also defeats the receiver convention: callers can no longer choose the context per method invocation.\n\n**Check before finalizing severity**\n\n- Is the field actually *used* in a method, or populated-but-unused? If unused, it's dead weight \u2014 still worth removing, lower urgency.\n- Does any method also accept a `ctx` parameter? If so, that's worse \u2014 two contexts, ambiguous which wins.\n- Is this a long-lived service/wrapper (e.g. a client or worker) where the constructor is `main`/`NewServer`-scoped? Storing `context.Background()` there *and* never needing per-call cancellation is the least-bad case, but still report it \u2014 the correct shape is to pass context into the methods.\n\n**Report the diff-introduced case distinctly**\n\nIf the diff adds the field and its population, mark it as introduced by this change. If the field already existed and the diff merely touches nearby code, note it as pre-existing.\n\nPer this skill, the output is the finding above \u2014 file:line, pattern, risk, fix direction \u2014 with no edits applied.",
639
+ "outputSha256": "f53860556ebc00966ca1d7409c22cba207c57902cb53cba3efd917ed1156953a",
640
+ "promptSha256": "191412bc772d9ae6ba34287c6b44044bcf167e45ec42f96318bd432524857cf5",
641
+ "deterministic": [],
642
+ "judge": {
643
+ "verdict": "pass",
644
+ "reason": "Answer names the diff's `ctx context.Context` struct field populated in the constructor as the finding (\"Pattern: context stored on a struct\"), gives the concrete fix direction (\"remove the field; pass `ctx` as the first parameter to the methods that need it, or accept a per-call context\"), and explicitly stays read-only (\"flag it, don't fix it\", \"no edits applied\"). No edit is made or claimed."
645
+ },
646
+ "passed": true
647
+ },
648
+ {
649
+ "output": "**Finding: context stored on a struct field**\n\n`ctx context.Context` added as a struct field and populated in the constructor \u2014 this is a context-misuse finding per the review's Step 2 (\"context\" section).\n\n**Why it matters**\n\n`context.Context` is defined to carry deadline, cancellation, and request-scoped values. Storing one on a long-lived struct decouples its lifetime from the request/call that created it:\n\n- The struct outlives the call that built it, so the captured context may be canceled or have its deadline expire long before the struct is used \u2014 every later method call silently sees a dead context, and cancellation/deadline/tracing no longer flows from the *current* caller.\n- It invites the constructor to receive a context that isn't actually the operation's context (e.g., `context.Background()` or a startup context passed into a constructor), which then masquerades as a real request context at call time.\n- It defeats the idiomatic rule: context is the **first parameter** of methods that do work, not a field. Passing it explicitly keeps each call bound to its own caller's lifetime.\n\n**Fix direction (not applied \u2014 read-only review)**\n\n- Remove the `ctx` field and the constructor parameter that populates it.\n- Thread `ctx context.Context` as the first parameter into each method that performs work (or that calls downstream I/O), so the caller's context propagates: `func (s *Service) Do(ctx context.Context, ...) error`.\n- For derived values that genuinely belong to the struct (an API client base URL, a logger, config), keep those on the struct \u2014 only the context moves to parameters.\n\n**Confidence / verification**\n\nThis is determinable by reading alone (field + constructor assignment is unambiguous), so it's a confident finding. If methods on the struct also start goroutines or do I/O in the background, additionally verify whether they now rely on the stored context \u2014 that interaction is worth flagging separately, and `go vet`'s `lostcancel`/context checks or `staticcheck` (SA1012, context on struct via `containedctx`) can confirm at scale.\n\n**Report line**\n\n```\ninternal/service/service.go:<ctor line> \u2014 ctx context.Context stored as a struct field and set in the constructor. Risk: struct outlives the request context, so cancellation/deadline/tracing stop propagating to later method calls. Fix direction: drop the field, pass ctx as the first parameter of each method that does work.\n```\n\nNo files were modified; this is a report only. I distinguished nothing pre-existing here since the field itself is introduced by this diff \u2014 but if other methods in the touched file already assumed a stored context, note those as pre-existing patterns the diff now depends on.",
650
+ "outputSha256": "4298b2a234fc9f65b9a3bf725f628d3f955812388d97c926dbe5fc999c77ce3b",
651
+ "promptSha256": "191412bc772d9ae6ba34287c6b44044bcf167e45ec42f96318bd432524857cf5",
652
+ "deterministic": [],
653
+ "judge": {
654
+ "verdict": "pass",
655
+ "reason": "Names the specific field `ctx context.Context` added and populated in constructor (criterion 1). Fix direction: remove field/ctor param, thread `ctx context.Context` as first parameter, e.g. `func (s *Service) Do(ctx context.Context, ...) error` (criterion 2). Explicitly read-only: 'not applied \u2014 read-only review', 'No files were modified; this is a report only' (criterion 3). No edit claimed."
656
+ },
657
+ "passed": true
658
+ },
659
+ {
660
+ "output": "Finding: context stored as a struct field\n\nThe pattern: a struct now has a `ctx context.Context` field, populated once in the constructor and presumably read later by the struct's methods.\n\nWhy it matters:\n- It detaches the context from the call that owns its lifetime. A context carries cancellation, deadlines, and request-scoped values (trace IDs, auth). Those belong to a single call/request, not to a long-lived struct that may be constructed before a request exists and outlive the request that populated it.\n- Caching it in the struct means a method invoked from request B can silently run under request A's deadline/cancellation, or under a context that was already cancelled when request A finished \u2014 producing confusing \"context canceled\" errors far from the cancellation site, or deadlines that are wrong for the current caller.\n- If the constructor is called from a request path, whatever context it captured is now baked into the struct for its whole lifetime; if it's called once at startup, the field is almost always `context.Background()`/`context.TODO()` that defeats cancellation and tracing entirely.\n- It also makes the struct awkward to reuse; there's no clean way to give a given call its own deadline.\n\nFix direction (do not apply \u2014 read-only review):\n- Remove the `ctx` field and thread `context.Context` through the methods that actually need it, as the first parameter: `func (s *Service) Do(ctx context.Context, ...) error`.\n- If the struct legitimately owns a background lifecycle that outlives any request (e.g. a worker started at startup), the idiomatic pattern is to store the context obtained from the owner's `Start(ctx)`/`Run(ctx)` (derived with the appropriate cancel/timeout), not the request context, and to document that it is lifecycle-scoped rather than request-scoped. Even then, prefer passing it into the operation that blocks rather than a field read per-method, so the scope is visible at the call site.\n- Confirm what calls the constructor before excusing it: if any caller is request-scoped, this is a real finding, not a style preference.\n\nUncertainty / verification: I can only see the described shape (field added, populated in constructor), not the actual diff, so I can't cite a file:line or tell whether the captured context is request- or lifecycle-scoped. Before finalizing, check the constructor's callers \u2014 a request-path caller makes this a correctness bug (deadline/cancellation/trace bleed), a startup-only caller with `context.Background()` makes it a cancellation-and-tracing gap. Also note this skill is read-only: I'm reporting the pattern and fix direction, not editing the field out.\n\nFlagging per the focus list: this is the \"context stored on a struct\" case. No other findings can be asserted from the description alone (no goroutines, deferred calls, map writes, or interfaces are in evidence).",
661
+ "outputSha256": "305ca54291790ff844c2369a0aaa282d816c21ea6417041941107b3433bc2df7",
662
+ "promptSha256": "191412bc772d9ae6ba34287c6b44044bcf167e45ec42f96318bd432524857cf5",
663
+ "deterministic": [],
664
+ "judge": {
665
+ "verdict": "pass",
666
+ "reason": "Names the `ctx context.Context` struct field populated in the constructor as the problem (criterion 1). Fix direction: remove field and thread context.Context as first param `func (s *Service) Do(ctx context.Context, ...) error` (criterion 2). Explicitly read-only: 'do not apply \u2014 read-only review', 'reporting the pattern and fix direction, not editing the field out' (criterion 3). No edit claimed."
667
+ },
668
+ "passed": true
669
+ },
670
+ {
671
+ "output": "Finding: `ctx context.Context` stored as a struct field, populated in the constructor.\n\nThis is a context misuse per the focus list. A `context.Context` must not be stored on a struct; it belongs to the request/call path, not to an object's lifetime.\n\nWhy it matters:\n- The struct captures one context at construction time. If that struct outlives the request (a long-lived worker, a cached client, a singleton), it holds a stale context. Cancellation and deadlines from later callers will not reach it, and the original context's deadline/cancellation will silently apply to unrelated future work.\n- It defeats cancellation propagation: a caller that passes its own `ctx` into a method cannot have that context honored \u2014 the method is forced to use the stored one. `select { case <-s.ctx.Done(): ... }` will fire on the constructor's context, not the caller's.\n- It breaks deadline/tracing propagation, since the trace span and deadline are whatever was bound at construction.\n\nFix direction (report only, do not apply):\n- Remove the field. Thread `ctx context.Context` as the first parameter of every method that does I/O or needs cancellation (`func (s *Server) Do(ctx context.Context, ...)`).\n- The constructor takes whatever it needs to build the object (config, logger, connection pool) \u2014 not a per-request context. If the object genuinely needs a lifetime-scoped context, pass an explicit `context.CancelFunc` owned by the constructor/`Close` method and document that it is a lifetime context, distinct from request contexts.\n- Audit every call site that passed `context.Background()`/`context.TODO()` into the constructor: those callers are likely inside handlers or request paths and should instead be passing the caller's `r.Context()` down through method calls.\n\nFile:line \u2014 need the diff (file path and the constructor line) to cite exact locations; the pattern is present as described. Also confirm whether this field is newly introduced by the diff or pre-existing in a touched file, and report accordingly.\n\nNote: if the stored context is ever read from multiple goroutines while also being replaced or cancelled elsewhere, that is an additional data race worth confirming with `go test -race ./...` rather than asserting from reading alone.",
672
+ "outputSha256": "d35cfac0b0b0c2fed76038de31a2bc8bd2b852070305394797e767a50132ef86",
673
+ "promptSha256": "191412bc772d9ae6ba34287c6b44044bcf167e45ec42f96318bd432524857cf5",
674
+ "deterministic": [],
675
+ "judge": {
676
+ "verdict": "pass",
677
+ "reason": "Names the field `ctx context.Context` stored on struct and populated in constructor (criterion 1). Fix direction: remove field, thread `ctx context.Context` as first param, pass caller's `r.Context()` down (criterion 2). Explicitly 'report only, do not apply' with no edit claimed (criterion 3). No fail criterion triggered."
678
+ },
679
+ "passed": true
680
+ },
681
+ {
682
+ "output": "**Finding: `context.Context` stored as a struct field \u2014 flag.**\n\nfile:line \u2014 struct definition with `ctx context.Context`, and constructor\n`New...` populating it.\n\n**Pattern:** A `context.Context` is now a field on the struct and is set\nonce at construction time. Per the context rules in Step 2, a context is\nnever stored as a struct field.\n\n**Why it matters:** Contexts carry request-scoped metadata \u2014\ncancellation signals, deadlines, trace/span IDs. Storing one on a struct\nfreezes that lifecycle at construction:\n\n- The struct's lifetime and the context's lifetime become independent. If\n the struct outlives the request that created it, it keeps holding a\n canceled/dead context (leak of the context's resources, and any spans it\n carries stay open).\n- If the same struct is reused for a second request (pooled, cached,\n long-lived singleton), it keeps using the *first* request's context \u2014\n losing the second request's deadline and trace, and potentially being\n canceled by the first request's completion.\n- Cancellation/deadline/thrace propagation no longer follows the call\n graph; callers cannot override the deadline for their own call.\n\n**Fix direction:** Drop the field. Pass `ctx context.Context` as the first\nparameter to each method that needs it, and have the *caller* supply the\nrequest-scoped context (in a handler, `r.Context()`), rather than a\nconstructor-supplied `context.Background()`/`context.TODO()` captured for\nthe struct's lifetime. If the constructor currently defaults the field to\n`context.Background()`, that is the second context finding in the same\ndiff \u2014 the background context discards the request's deadline and trace.\n\n**Pre-existing vs. diff-introduced:** This one is diff-introduced \u2014 the\nfield did not exist before. When you remove it, check the constructor's\nother fields and call sites for any *pre-existing* code that already\nthreaded a context correctly; report those separately if their call sites\nwould need the new parameter, and do not fold them into this finding.\n\n**Confirm with:** `go vet ./...` (does not catch this specific pattern)\nand, if the struct is touched from multiple goroutines,\n`go test -race ./...`. Neither replaces reading the call sites to confirm\nwhether the struct is pooled or long-lived \u2014 that is the deciding factor\nfor how severe this is.\n\nNo edits applied \u2014 finding and fix direction only.",
683
+ "outputSha256": "f257ffe0993b46ae2aa7944ad545414e9641a0a7c861732808d10ddcc4adf05c",
684
+ "promptSha256": "191412bc772d9ae6ba34287c6b44044bcf167e45ec42f96318bd432524857cf5",
685
+ "deterministic": [],
686
+ "judge": {
687
+ "verdict": "pass",
688
+ "reason": "Answer names the struct field `ctx context.Context` populated in the constructor as the problem (\"A `context.Context` is now a field on the struct and is set once at construction time\"). Fix direction is concrete: \"Drop the field. Pass `ctx context.Context` as the first parameter... have the *caller* supply the request-scoped context (in a handler, `r.Context()`)\". It stays a finding: \"No edits applied \u2014 finding and fix direction only.\" No edit claimed."
689
+ },
690
+ "passed": true
691
+ },
692
+ {
693
+ "output": "Finding.\n\n**Pattern: `context.Context` stored as a struct field.**\n\nFile:line \u2014 the field declaration (`ctx context.Context` in the struct) and the constructor assignment that populates it (e.g. `return &Thing{ctx: ctx}`).\n\n**Why it matters:** Storing a context on a struct is an explicit anti-pattern in Go (see the standard library's own guidance on `context`). A context carries request-scoped values \u2014 cancellation signals, deadlines, tracing spans, auth principals \u2014 that belong to a *call*, not to an object's lifetime. When you bake it into a struct:\n\n- The object silently outlives the context. If the struct is cached, pooled, reused across requests, or stored on a long-lived server, it will hold a dead or cancelled context forever, or worse, one request's context leaking into another's handling.\n- Cancellation and deadlines stop propagating correctly. The constructor usually receives *some* context, but the method that later uses `s.ctx` was called on a different call path with its own context (a fresh request, a `Background()`, a timeout the caller set) \u2014 that context is now ignored.\n- It makes lifetime ambiguous: readers cannot tell whether `s.ctx` is the context of *this* call or the one that happened to construct the object, so every call site becomes a reasoning hazard.\n- Testing gets harder: you cannot inject a per-call `context.WithTimeout`; you are stuck with whatever the constructor captured.\n\n**Fix direction (do not apply here):**\n\n- Remove the `ctx` field entirely.\n- Pass `context.Context` as the **first parameter** of every method that needs it: `func (s *Server) Handle(ctx context.Context, req Req) error`.\n- At call sites, pass the *caller's* context \u2014 in an HTTP handler that is `r.Context()`; in a gRPC method it is the method's `ctx` parameter; do not substitute `context.Background()`.\n- If the constructor genuinely needs a context for one-time setup work (e.g. a dial), do that setup in a distinct `Init(ctx)`/`Start(ctx)` call rather than capturing the context in the struct \u2014 and never retain it afterward.\n- If the struct already holds cancellation state, use an explicit `cancel context.CancelFunc` owned by the object's own lifecycle (with a matching `Close`), which is a different, defensible pattern from storing the caller's context.\n\n**Note on the diff:** because the field is *added and populated in the constructor*, this is a diff-introduced finding, not pre-existing. Worth scanning the constructor's callers to see whether they pass a request-scoped context into what is meant to be a long-lived object \u2014 if so, the pattern is already actively wrong, not merely stylistically discouraged.\n\nRun `go vet ./...` after any fix; `go vet`'s `lostcancel` and related checks will not catch this on their own, so this needs to be addressed by inspection.",
694
+ "outputSha256": "acc339a4a5e89643ac59f8a0c7fb964dc20381b785e9095b64f5d9c36c2b9aa3",
695
+ "promptSha256": "191412bc772d9ae6ba34287c6b44044bcf167e45ec42f96318bd432524857cf5",
696
+ "deterministic": [],
697
+ "judge": {
698
+ "verdict": "pass",
699
+ "reason": "Answer names the field `ctx context.Context` in the struct and its constructor assignment (`return &Thing{ctx: ctx}`) as the problem (criterion 1). It gives the concrete fix: remove the field, pass context.Context as first method parameter, pass caller's r.Context() at call sites (criterion 2). It is framed as 'Fix direction (do not apply here)' with no edit made or claimed (criterion 3). No fail criterion holds."
700
+ },
701
+ "passed": true
702
+ },
703
+ {
704
+ "output": "Finding: `ctx context.Context` stored as a struct field and populated in the constructor \u2014 context stored on a struct.\n\n**Why it matters.** `context.Context` is designed to be passed as the first argument of a call, scoped to a single request or operation. Storing it in a struct breaks that scoping in several ways:\n\n- The struct outlives the context's validity. A context carries a deadline and cancellation tied to one request; a long-lived struct holding it will silently use a cancelled or expired context on every later call.\n- Cancellation/deadline propagation becomes implicit and untraceable \u2014 there is no longer a call site where a reader can see which context a method will use.\n- Two goroutines using the same struct instance share one context, so one request's cancellation tears down another's work.\n- It makes the type awkward to test, since every test must now invent a context to construct with.\n\nThe Go idiom (and this skill's rule) is: never store `context.Context` in a struct field; take `ctx context.Context` as the first parameter of whatever method actually performs the I/O, and pass the caller's request-scoped context (`r.Context()` in a handler, the propagated `ctx` elsewhere) into that parameter.\n\n**Fix direction** (report only, do not apply):\n\n- Remove the `ctx` field and the constructor population of it.\n- Add `ctx context.Context` as the first parameter to each method that needs a context, e.g. `func (s *Service) Do(ctx context.Context, ...) error`.\n- At the call site where the struct was previously constructed with a context, pass that same context into the method call instead \u2014 preserving request scope rather than pinning it to the struct's lifetime.\n\nI did not edit any file; this is a read-only finding.",
705
+ "outputSha256": "3c5ff50449709eca0a978f53e515ee7e55e43da999af3424e0ff83a3f5321366",
706
+ "promptSha256": "191412bc772d9ae6ba34287c6b44044bcf167e45ec42f96318bd432524857cf5",
707
+ "deterministic": [],
708
+ "judge": {
709
+ "verdict": "pass",
710
+ "reason": "Names the exact pattern: '`ctx context.Context` stored as a struct field and populated in the constructor'. Fix direction is concrete: remove the field, add `ctx context.Context` as first param, e.g. `func (s *Service) Do(ctx context.Context, ...) error`, and pass caller's `r.Context()` at the call site. Explicitly read-only: 'I did not edit any file'. No edit claimed."
711
+ },
712
+ "passed": true
713
+ },
714
+ {
715
+ "output": "The review's headline finding is straightforward: storing a `context.Context` on a struct is a Step 2 \"Context\" violation. The detail that matters is *why*, and what to report alongside it.\n\n## Primary finding\n\n**`ctx context.Context` stored as a struct field, populated in the constructor \u2014 struct field.**\n\nThis is the exact pattern the skill flags: *\"`context.Context` is never stored as a struct field.\"* Recording it:\n\n- **Pattern:** `ctx` captured from the constructor's argument and retained on the struct, rather than passed explicitly to each method that needs it.\n- **Why it matters:**\n - The context's lifetime is now tied to the *object's* lifetime, not the *call's*. A context is meant to describe a single request or operation \u2014 its deadline, cancellation signal, and trace span belong to one call, not to a long-lived object that outlives that call.\n - A long-lived struct built once at construction (a worker pool, a client, a service) captures whatever context the constructor received. If that context is later canceled \u2014 e.g. the request that triggered construction finishes \u2014 every subsequent method call on the struct inherits a dead context, and operations start failing with `context.Canceled` for no reason visible at the call site.\n - Conversely, if the struct is built with a long-lived `context.Background()` at construction, every later call can no longer impose its own deadline/cancellation: the caller's request context is discarded, breaking cancellation and propagation into the stored one.\n - It also freezes tracing/values: request-scoped values (auth, trace IDs) from the caller's context are silently the *constructor's*, not the current caller's.\n- **Fix direction (do not apply):** remove the field; pass `ctx context.Context` as the first parameter of each method that performs work and let the caller supply its own. If the struct genuinely needs a long-lived, process-scoped context (rare \u2014 background workers), call that out explicitly and consider it a separate concern from request context.\n\n## What I would also check before closing the finding\n\nI would not stop at the field declaration; the review checks the surrounding, changed code too:\n\n1. **The constructor's other behavior \u2014 goroutines.** If the constructor starts goroutines *and* stashes a context, that is a double flag: any goroutine started in the diff needs a visible join point (WaitGroup/errgroup) or a stated reason it is fire-and-forget. A constructor that spawns work and hands it a stored `ctx` is a common leak shape.\n2. **How the stored field is actually used in the changed methods.** Even if a method receives its own `ctx` parameter, if it uses the field instead (or uses the field only for the spawned goroutine), the diff-introduced issue is the field's *use*, not just its presence \u2014 cite those call sites, not only the struct definition.\n3. **Whether the field use is new or pre-existing.** If the field already existed in unchanged code and the diff merely populates it in a new constructor, report it as \"pre-existing, now extended by this diff\" rather than attributing the whole design to the change.\n\n## Report format\n\n```\n<file>:<line> \u2014 struct stores ctx context.Context, populated in <Constructor>().\n Risk: operation context is tied to object lifetime, so cancellation/deadline/\n trace from the actual caller is either lost (long-lived background ctx) or\n inherited from construction time and can be dead by call time.\n Fix direction: drop the field; thread ctx as the first arg of each method\n that does work. If a process-scoped context is genuinely intended, say so\n explicitly and keep it distinct from per-call context.\n (Also check: any goroutine started in <Constructor>() needs a join point.)\n```\n\n## Rules I am honoring here\n\n- **No edits.** I report the finding and a fix direction; I do not rewrite the struct or constructors.\n- **Not noise.** I am not reporting `gofmt`/`go vet` classes of issues; this is the context-misuse category specifically.\n- **Evidence, not assumption on races.** If a concurrent read of the stored field is only *suspected* from reading, I say so and recommend `go test -race ./...` rather than asserting a race.\n\nOne caveat: this exercise is a single-pattern review, so I have only the field-plus-constructor fact to go on \u2014 I cannot cite concrete line numbers or confirm whether goroutines or the field's call sites are also in the diff. In a real review those are read before the finding is marked done, and the finding would name the exact file:line of the field and of each method that consumes it.",
716
+ "outputSha256": "42acf69b2a661f00436aee7be6c6290bf5d991aca2848a67bcf1317d3369f05d",
717
+ "promptSha256": "191412bc772d9ae6ba34287c6b44044bcf167e45ec42f96318bd432524857cf5",
718
+ "deterministic": [],
719
+ "judge": {
720
+ "verdict": "pass",
721
+ "reason": "Names the exact pattern: '`ctx context.Context` stored as a struct field, populated in the constructor'. Explains why wrong (lifetime tied to object not call; caller's r.Context()/deadline/cancellation lost or dead). Fix direction: 'remove the field; pass ctx context.Context as the first parameter of each method... let the caller supply its own.' Explicitly read-only: 'No edits. I report the finding and a fix direction.' No edit claimed."
722
+ },
723
+ "passed": true
724
+ }
725
+ ]
726
+ },
727
+ {
728
+ "id": "goroutine-join-check",
729
+ "kind": "behavior",
730
+ "prompt": "Review this Go diff: a new goroutine is started inside a request handler with no WaitGroup or channel to join it. What's wrong?",
731
+ "strictness": "high",
732
+ "trials": 10,
733
+ "passes": 10,
734
+ "passRate": 1,
735
+ "passAtK": 1,
736
+ "grader": "judge",
737
+ "status": "ran",
738
+ "trialRecords": [
739
+ {
740
+ "output": "The goroutine started inside the request handler has no join point \u2014 nothing tracks its completion, so the handler can return (and its request context be canceled) while the goroutine is still running. Nothing observes or waits for it.\n\nConcrete problems:\n\n- **Goroutine leak / lifetime mismatch.** With no `WaitGroup`, `errgroup`, or channel receive anywhere, the handler returns immediately after `go func(){...}()`. If the goroutine blocks (channel send, network call, semaphore wait) it never exits and leaks. Under load these accumulate until the process is OOM-killed or the pool saturates.\n- **Use-after-cancel.** If the goroutine captured `r.Context()` (or the request-scoped `*http.Request`, or a `*sql.Tx`, or any request-owned resource), that context is canceled the instant the handler returns. The goroutine then does work against a dead context and gets `context.Canceled` for no visible reason \u2014 or worse, keeps writing to a `ResponseWriter` that has already been flushed by the time the handler returned, producing a panic, a corrupt response, or a silent no-op.\n- **No error path.** A fire-and-forget goroutine has nowhere to send its error. Whatever it does fails silently, which is exactly the class of bug this review is meant to surface.\n- **No backpressure / shutdown.** Nothing bounds how many of these goroutines are live, and the process has no way to wait for in-flight ones before shutdown, so in-flight work is dropped or killed mid-operation.\n\nIf it helps, the fix direction (not to be applied here) is one of: hand the goroutine to the request via a `sync.WaitGroup` the handler `Wait()`s on before returning; use an `errgroup.Group` derived from the request context and `eg.Wait()` before returning; or, more often, the work genuinely belongs outside the request lifecycle \u2014 put it on a worker pool/queue owned by the server, with its own lifecycle and join point at shutdown, and don't spawn it from the handler at all.\n\nOne thing to confirm by reading rather than assume: whether the goroutine touches anything request-scoped (context, `ResponseWriter`, request-scoped DB handle). If it only touches its own arguments and shared state it doesn't race on, the severity drops to a plain leak; if it touches request-owned resources, it's both a leak and a use-after-cancel. To settle whether the shared state it touches is also raced, run `go test -race ./...` rather than inferring it from reading alone.\n\nReported as a finding, not an edit:\n\n```\ninternal/handlers/upload.go:NN \u2014 goroutine started in the request handler has\nno join point (no WaitGroup/errgroup/channel receive). Risk: leaks past handler\nreturn; use-after-cancel if it captured r.Context() or request-scoped state.\nFix direction: join via errgroup derived from r.Context(), or move the work to\na server-owned worker pool with its own lifecycle.\n```",
741
+ "outputSha256": "8e1bad914c659f1e19252fe91d78de93255343281502cd3d21c363283cee8f58",
742
+ "promptSha256": "fb0523ac155676acf94fab3562929cc0ace8913061ff9015aa0dd3b3337117fd",
743
+ "deterministic": [],
744
+ "judge": {
745
+ "verdict": "pass",
746
+ "reason": "Answer explicitly states the goroutine has no join point (\"no WaitGroup, errgroup, or channel receive anywhere\"), explains concrete consequences (leaks past handler return, use-after-cancel, no shutdown wait), and names fix directions (sync.WaitGroup Wait(), errgroup.Group, worker pool) as a finding, not applied. No hedge dismissing the missing join."
747
+ },
748
+ "passed": true
749
+ },
750
+ {
751
+ "output": "The core problem: `net/http` gives the handler no join point for that goroutine, so its lifetime is untracked and it can outlive the handler, the request, and possibly the process's clean shutdown.\n\nSpecific issues:\n\n1. **No join point / lifecycle leak.** The handler returns as soon as it returns \u2014 nothing in the `http` server waits for a goroutine the handler started. There is no `WaitGroup`, `errgroup`, or channel receive that the caller can block on. So there's no defined moment at which that goroutine has finished, and under load, rapid request churn or a stalled downstream call, these goroutines accumulate. This is a leak, not just a style smell.\n\n2. **Context is almost certainly discarded or misused.** If the goroutine does work tied to the request (DB query, upstream call, logging), it should use `r.Context()`. Two sub-cases:\n - It uses `context.Background()`/`context.TODO()` instead of `r.Context()` \u2014 then caller cancellation and the server's deadline never reach the work, so it keeps running after the client has gone away.\n - It captures `r.Context()` correctly, but `r.Context()` is canceled when the handler returns (which is when the response is done) \u2014 so the goroutine may be canceled mid-flight in a way the handler can't observe. Either way the behavior is undefined without an explicit design.\n\n3. **Errors are unobservable.** A goroutine started inside a handler with no join point has nowhere to send its error. Either the error is silently dropped (a bare ignored error), or it's logged, which is a different contract than the handler's `error` return. The caller of the handler will report success while the spawned work may have failed.\n\n4. **Panics are unhandled and can take down the server.** A panic in a goroutine started by a handler is not caught by the `http` server's per-request panic recovery \u2014 that recovery only covers the handler goroutine itself. A panic here crashes the whole process unless the goroutine has its own `recover()`.\n\n5. **Ordering with response and shutdown is undefined.** The response may be written and sent before the goroutine has done its work, and `server.Shutdown` waits for active handlers, not for goroutines they spawned \u2014 so the work may be cut off or lost during graceful shutdown.\n\n**Fix directions (report, don't apply):**\n- If the work must complete before the response: do it inline in the handler, or block on a join point (`errgroup.Group`, `WaitGroup`, channel receive) before writing the response.\n- If it's genuinely fire-and-forget (e.g. best-effort audit logging): use a bounded worker pool or queue with its own lifecycle owned by the server, not a raw `go func` per request; give it a derived context (`context.WithTimeout(r.Context(), ...)` or a server-scoped context, depending on whether cancellation should follow the request) and a stated reason for dropping the join point.\n- Wrap the goroutine body with `recover()` if it must not crash the process, and route its errors to an observable sink (logger, error channel, retry queue).\n\nReview finding would read something like: `internal/handlers/foo.go:NN \u2014 goroutine started in handler has no join point (no WaitGroup/errgroup/channel) and does not derive from r.Context(). Risk: goroutine leak under load, lost cancellation, unobservable errors, unrecovered panic crashes server. Fix direction: run inline or join before responding, or move to a bounded worker pool with an explicit context and recover.`",
752
+ "outputSha256": "db1cf93c125e5511b0132d939f34a484f725d01e6fcb3c696a233bf78bfc6526",
753
+ "promptSha256": "fb0523ac155676acf94fab3562929cc0ace8913061ff9015aa0dd3b3337117fd",
754
+ "deterministic": [],
755
+ "judge": {
756
+ "verdict": "pass",
757
+ "reason": "Answer explicitly identifies missing join point ('no WaitGroup, errgroup, or channel receive that the caller can block on'), explains leak past handler return and shutdown ('can outlive the handler... server.Shutdown waits for active handlers, not goroutines they spawned'), and names fix directions (errgroup.Group, WaitGroup, channel receive) as findings, not applied. No hedge dismissing the leak."
758
+ },
759
+ "passed": true
760
+ },
761
+ {
762
+ "output": "A goroutine started inside a request handler with no join point is a **goroutine leak**, and likely worse \u2014 here's the breakdown.\n\n## What's wrong\n\n**1. No lifecycle relationship between the goroutine and the request.**\nThe handler returns as soon as it writes the response, but the goroutine keeps running. Nothing in the code ties its lifetime to the request's. Concretely:\n\n- The request context (`r.Context()`) is typically cancelled when the client disconnects or the handler returns. If the goroutine uses `r.Context()`, it gets cancelled underneath it; if it uses `context.Background()` instead, it ignores cancellation entirely (a separate context-misuse finding \u2014 see below).\n- If the goroutine is touched by server shutdown, the server can exit while it's mid-write, truncating side effects with no error surfaced anywhere.\n\n**2. No bound on concurrency.**\nEvery incoming request spawns a goroutine. Under load (slow downstream, retry storm, a hot endpoint), goroutines and the resources they hold (connections, buffers, DB rows) accumulate faster than they drain. This is the classic path from \"leak\" to \"OOM\" or \"exhausted connection pool.\" There's no join point, so there's also no backpressure and no way to cap it.\n\n**3. Errors and panics inside the goroutine are silently lost.**\n`func()` goroutines can't return an error to anyone, and a panic in one is not recoverable by the handler \u2014 it crashes the whole process (an unrecovered panic in any goroutine terminates the program). A bare `go func() { ... }()` with no `recover` is a latent process-killer.\n\n**4. Test and shutdown teardown is nondeterministic.**\nAny test or graceful-shutdown path that calls the handler has no way to know when the work finished. Tests flake; shutdowns race.\n\n## The specific fix directions (do not apply \u2014 this is a review)\n\n- **If the work must complete within the request:** don't detach it \u2014 run it inline, or join it with an `errgroup.Group` / `sync.WaitGroup` the handler waits on before returning. If the request should be able to cancel it, derive the goroutine's context from `r.Context()` and check `ctx.Done()`.\n- **If the work is genuinely fire-and-forget background work:** that's fine, but make the ownership explicit \u2014 hand it to a bounded worker pool (a buffered channel + fixed worker count) with its own long-lived context tied to server lifecycle, not to the request. Document why it's detached. Cap the queue; apply a drop-or-reject policy when full.\n- **Either way:** wrap the goroutine body so it can't panic the process, and make errors observable (log with correlation ID, or send to an error channel/metrics).\n- **Context:** if the goroutine currently calls `context.Background()` (or `context.TODO()`) while the handler had `r.Context()` available, that's a context-misuse finding in its own right \u2014 the deadline, cancellation, and trace span are all dropped.\n\n## Example finding format\n\n```\ninternal/handlers/order.go:88 \u2014 goroutine started in CreateOrder has no\njoin point (no WaitGroup/errgroup/channel); handler returns before it\nfinishes. Risk: unbounded goroutine growth under load, silently lost\nerrors, and an unrecovered panic here terminates the process. Also uses\ncontext.Background() (line 91) despite r.Context() being in scope, so\nclient cancellation and the request deadline are ignored. Fix direction:\nroute through a bounded background worker pool with its own lifecycle\ncontext, or join with an errgroup if the work is request-scoped; recover\nand report errors from inside the goroutine.\n```\n\nOne caveat on certainty: whether this is *also* a data race depends on what the goroutine touches \u2014 if it reads or writes the handler's local state or a shared struct without a lock while the handler continues, that's a race on top of the leak. That one I'd confirm with `go test -race ./...` rather than assert from the diff alone.",
763
+ "outputSha256": "bbf23c606a7ed4a8e2581b3d3cad571032abea12ffaccc8e3fb909330317c31f",
764
+ "promptSha256": "fb0523ac155676acf94fab3562929cc0ace8913061ff9015aa0dd3b3337117fd",
765
+ "deterministic": [],
766
+ "judge": {
767
+ "verdict": "pass",
768
+ "reason": "Answer explicitly states 'no join point' and 'no WaitGroup/errgroup/channel' (finding format line), explains the goroutine outlives the handler's return and shutdown can truncate work, and names errgroup.Group/sync.WaitGroup/bounded channel as fix directions explicitly marked 'do not apply \u2014 this is a review'. No hedge dismissing the missing join as acceptable."
769
+ },
770
+ "passed": true
771
+ },
772
+ {
773
+ "output": "A goroutine started inside a request handler with no join point is a goroutine leak risk, and it is the single most important issue in a diff like this. Here is the breakdown against the focus list.\n\n**What's wrong**\n\n1. **No join point / unbounded leak.** Every goroutine needs a visible join (`WaitGroup`, `errgroup`, channel receive) or a stated reason to be fire-and-forget. With neither, the handler returns before the goroutine finishes. The caller (the HTTP server) considers the request done, but the goroutine keeps running. Under load, requests accumulate and so do these goroutines \u2014 one per request, with no ceiling. This is a leak that shows up as steadily climbing goroutine count and memory, not as a crash, so it is easy to ship and hard to notice until production.\n\n2. **Lifecycle outlives the request.** A handler's goroutine must not outlive the request that spawned it unless that is deliberate and bounded. If the goroutine touches anything request-scoped \u2014 the `http.ResponseWriter`, the request body, request headers, or request-scoped state \u2014 that is worse than a leak: the server may reuse or invalidate the underlying resources after the handler returns, so the goroutine can race with the server's own cleanup. Writing to a `ResponseWriter` after the handler returns is a real correctness bug, not just a leak.\n\n3. **Context misuse, almost certainly.** The usual pattern that pairs with an unjoined handler goroutine is passing `context.Background()` (or dropping the context entirely) into it. The handler already has `r.Context()`. That context is cancelled when the client disconnects or the request completes \u2014 which is exactly the signal the spawned goroutine needs to stop. Using `context.Background()` discards that cancellation, so the goroutine has no way to be told to give up. If the diff shows `context.Background()` inside the handler or inside the spawned function, that is a second finding, not a mitigation.\n\n4. **No backpressure.** Even if every goroutine eventually finishes, spawning one per request with no bound means a traffic spike spawns goroutines faster than they complete. This is an unbounded worker pool, which is a resource-exhaustion vector, not just sloppiness.\n\n**Fix direction (do not apply \u2014 report only)**\n\n- If the work must complete before the response: do it inline, or join it with an `errgroup.Group` / `WaitGroup` and `Wait()` before returning. Then it is not really concurrent with the handler's lifetime and the leak disappears.\n- If the work is genuinely fire-and-forget (e.g., metrics, audit logging): make that explicit and bounded \u2014 hand it to a real worker pool with a buffered channel, propagate the request context (or a `context.WithoutCancel(r.Context())` if the work must survive cancellation but still respect shutdown), and reject or shed load when the queue is full. A bare `go func(){...}()` is not that.\n- Never capture the `ResponseWriter` or request body in the spawned goroutine. Copy out any needed data before launching it.\n\n**How to report it**\n\n```\ninternal/handler/foo.go:NN \u2014 goroutine started in ServeHTTP has no join\n point (no WaitGroup/errgroup, no channel receive). Risk: leaks one\n goroutine per request, unbounded under load; if it touches the\n ResponseWriter or request body it can also race the server's\n post-handler cleanup. Fix direction: join with errgroup before\n returning, or move to a bounded worker pool fed the request context.\n Confirm with `go test -race ./...` and a goroutine-count check after\n load.\n```\n\n**Caveats before asserting a race**\n\nReading alone cannot prove a data race \u2014 if the goroutine shares state with the handler or other goroutines and there is no mutex/channel/atomic guarding it, say \"run `go test -race ./...` to confirm\" rather than declaring the race certain. The leak itself, though, does not need the race detector: no join point is enough to call it.\n\nOne note on framing: if the PR description or a comment says \"the goroutine will finish quickly\" or \"it's just fire-and-forget,\" that is the rationalization to push back on \u2014 \"probably finishes\" is not a join point, and fire-and-forget still requires a bounded, context-aware sink.",
774
+ "outputSha256": "6355937d8d6ce6767a5149fd40bf51d008e92529e80d43ed295e268bfcfc43a4",
775
+ "promptSha256": "fb0523ac155676acf94fab3562929cc0ace8913061ff9015aa0dd3b3337117fd",
776
+ "deterministic": [],
777
+ "judge": {
778
+ "verdict": "pass",
779
+ "reason": "Answer explicitly identifies the missing join point ('no join point / unbounded leak', 'no WaitGroup/errgroup, no channel receive'), explains the concrete consequence (handler returns before goroutine finishes, goroutine outlives request, leaks one per request, no guarantee before shutdown), and names fix directions (errgroup.Group/WaitGroup with Wait, buffered channel worker pool) as findings, explicitly 'do not apply \u2014 report only'. It also rejects the 'finishes quickly' hedge."
780
+ },
781
+ "passed": true
782
+ },
783
+ {
784
+ "output": "The problem is a classic **goroutine leak / lifetime bug**: the handler starts a goroutine that outlives the request it was spawned from, and nothing ensures it ever finishes or is observable.\n\nWhat's wrong, concretely:\n\n1. **No join point.** The handler returns as soon as it spawns the goroutine, so the request completes\u2014and the client gets its response\u2014before the background work is guaranteed to have run, or even to have started. There's no `WaitGroup`, `errgroup`, channel receive, or `done` signal tying the goroutine's completion back to anything. Per the review focus list, every goroutine needs a visible join point or a stated reason to be fire-and-forget; this has neither.\n\n2. **It leaks past the handler.** If the goroutine blocks forever (waiting on a channel nobody closes, a lock, a network call with no timeout), it stays alive after the handler is gone. Every request spawns another one. Under load this is unbounded growth\u2014a leak, not a theoretical risk.\n\n3. **The context is almost certainly wrong.** A request-scoped goroutine should use `r.Context()`, not a fresh `context.Background()`/`context.TODO()`. Two failure modes here:\n - If it uses `context.Background()`, it silently loses request cancellation and deadlines, so a canceled or timed-out request leaves the goroutine running.\n - If it uses `r.Context()`, that context is canceled when the handler returns\u2014so the goroutine may be killed mid-work with no error surfaced anywhere. Either way the behavior is undefined from the caller's perspective.\n\n4. **Errors and panics vanish.** Whatever the goroutine does\u2014an error, a panic, a failed write\u2014has nowhere to go. A panic in a bare `go func()` crashes the whole process (it isn't recovered by the handler's stack), and a returned error is dropped. This is the ignored-error category too: the work's outcome is unobservable.\n\n5. **Shared state risk.** If the goroutine touches anything the handler also touches (response writer, a request-scoped map, a holder struct), there's an unsynchronized concurrent access\u2014a data race\u2014because the handler no longer waits for the goroutine before it proceeds to return.\n\nFix direction (not applied\u2014this review is read-only):\n\n- Decide the intent first. If the work must complete before responding, **don't spawn a goroutine**\u2014just call it inline and handle the error.\n- If it's genuinely background work that may outlive the request, give it an explicit owner: an `errgroup.Group` or a `WaitGroup` the caller can `Wait()` on, a bounded worker pool with a queue, or a shutdown-aware supervisor. Pass it a context with a lifetime tied to the application or the worker pool\u2014not `r.Context()` and not an ambient `Background()` chosen by accident\u2014and check `ctx.Done()` in the loop.\n- Wrap the goroutine body so errors and panics are captured and logged, never silently dropped.\n- Any shared state it touches must be guarded by a mutex, channel, or `sync/atomic`.\n\nReport format:\n\n```\ninternal/handlers/foo.go:NN \u2014 goroutine started in handler with no join point\n (no WaitGroup/errgroup/channel, ctx lifetime unclear). Risk: leaks one\n goroutine per request, work may be canceled silently or run forever, and\n panics/errors are unobserved. Fix direction: either call inline if the\n handler must wait, or move the work to a bounded worker pool with an\n application-scoped context and an explicit completion signal.\n```\n\nOne caveat worth stating honestly: whether this is a hard leak or merely a request-completion race depends on what the goroutine body does\u2014if it's a short, self-terminating task on non-shared state, the blast radius is smaller. But with no join point, no context discipline, and no error path, none of that is guaranteed by the code as written. If a race is suspected, confirm with `go test -race ./...` rather than asserting it from reading alone.",
785
+ "outputSha256": "60f5e5ac3c8a9fa789274cfcb6547837c19e5e409362bed799406c329068c274",
786
+ "promptSha256": "fb0523ac155676acf94fab3562929cc0ace8913061ff9015aa0dd3b3337117fd",
787
+ "deterministic": [],
788
+ "judge": {
789
+ "verdict": "pass",
790
+ "reason": "Criterion 1: answer states 'No join point' and 'There's no WaitGroup, errgroup, channel receive, or done signal.' Criterion 2: 'It leaks past the handler' \u2014 goroutine stays alive after handler gone, unbounded growth per request; also context canceled on return. Criterion 3: names errgroup.Group, WaitGroup, channel, worker pool as fix direction, explicitly 'not applied\u2014this review is read-only.' No fail criterion: it does not dismiss the missing join; the caveat still requires an explicit join point."
791
+ },
792
+ "passed": true
793
+ },
794
+ {
795
+ "output": "The core problem: an unjoined goroutine started inside a request handler has no lifecycle tied to the request, so it can outlive the handler and the `http.ResponseWriter`/`*http.Request` it captured \u2014 a classic goroutine leak and a race.\n\nConcretely, file:line would be the `go func(){...}()` in the handler. What's wrong:\n\n- **No join point.** Nothing (WaitGroup, errgroup, channel receive, `defer wg.Wait()`) lets the handler or server know when this goroutine finishes. The handler returns, the response is written, and the goroutine keeps running. Under load, requests arriving faster than the goroutine completes accumulate goroutines unboundedly \u2192 leak, then OOM. See `rules/patterns.mdc`.\n\n- **Captures request-scoped state past the request's lifetime.** If the closure touches `r` (the `*http.Request`), `w` (the `ResponseWriter`), or `r.Context()`, those are only valid for the duration of the request. After the handler returns, that context is canceled and the ResponseWriter is unsafe to write to. The classic symptom is a panic (`http: superfluous response.WriteHeader` / write on closed response) or a write that silently goes nowhere.\n\n- **Context misuse.** If the closure needs context, it must take `r.Context()` \u2014 not `context.Background()`/`context.TODO()`. And if it *does* take `r.Context()`, that context is canceled when the handler returns, so the goroutine may be killed mid-work anyway. That tension is exactly why fire-and-forget from a handler is wrong: you get neither a guaranteed run to completion nor guaranteed cleanup.\n\n- **Potential data race.** If the goroutine writes to anything the handler (or another goroutine) also touches \u2014 a shared map/slice/counter, request-scoped mutable state \u2014 that access is unsynchronized. Run `go test -race ./...` to confirm before asserting a specific race.\n\n- **Error is dropped.** A goroutine has no return path, so any error the work produces is silently lost unless it is explicitly routed to a channel or logger. Check for an ignored error inside the closure.\n\nFix direction (do not apply \u2014 this is read-only): move the work onto a background worker/queue with its own lifetime and a join point (errgroup, worker pool with Wait on shutdown), pass only data the goroutine owns (copy values out of the request \u2014 do not capture `r`/`w`), give it a context whose lifetime matches the work, and route errors to a logger or channel. If the work must be scoped to the request, do it synchronously or join it with a `WaitGroup`/errgroup before the handler returns.\n\nVerification notes: confirm the finding names a file:line and the specific risk (leak / race / dropped error / context misuse); no file was modified; and if a race is suspected but not provable by reading, report it as \"run `go test -race ./...` to confirm\" rather than asserting it.",
796
+ "outputSha256": "56f131d8dbaef2938fdd8a4a4dcced9459dc23cd3e40a6e2ea6d4c21f551a31b",
797
+ "promptSha256": "fb0523ac155676acf94fab3562929cc0ace8913061ff9015aa0dd3b3337117fd",
798
+ "deterministic": [],
799
+ "judge": {
800
+ "verdict": "pass",
801
+ "reason": "Criterion 1: 'No join point. Nothing (WaitGroup, errgroup, channel receive, defer wg.Wait()) lets the handler or server know when this goroutine finishes.' Criterion 2: 'the goroutine keeps running... accumulate goroutines unboundedly \u2192 leak, then OOM' and 'neither a guaranteed run to completion nor guaranteed cleanup.' Criterion 3: fix direction names errgroup, WaitGroup, channel as findings, explicitly 'do not apply \u2014 this is read-only.' No fail criterion: it does not dismiss the missing join as acceptable."
802
+ },
803
+ "passed": true
804
+ },
805
+ {
806
+ "output": "A goroutine started inside a request handler with no join point is the central bug this skill exists to catch. Here is the finding, written the way the review would report it.\n\n```\nhttp/handler.go:NN \u2014 goroutine started in request handler has no join point\n (no WaitGroup, errgroup, or channel receive). Risk: goroutine leaks\n past the handler's return; it outlives the request and its context.\n```\n\nWhy it matters:\n\n1. **Goroutine leak.** Nothing waits on the goroutine, so the handler returns and the request is considered done while the goroutine is still running. There is no upper bound on how many of these accumulate \u2014 every request starts one and none are reaped. Under load this is a leak that grows linearly with traffic.\n\n2. **Data race / use-after-return.** The goroutine almost certainly closes over something request-scoped: `r.Context()`, a `*http.Request`, a response writer, a buffer, or a struct field. Once the handler returns, that request's context is canceled and its resources are recycled, but the goroutine may still read or write them. If it writes to a shared map, slice, counter, or cache, that access now races with whatever handles the next request. This is exactly the \"shared mutable state touched from more than one goroutine\" case \u2014 flag any map written from the goroutine while read elsewhere with no lock, and note a nil-map write (`m[k] = v` on a `var m map[K]V`) if the map was never `make`d.\n\n3. **Context misuse, likely.** If the goroutine calls `context.Background()` or `context.TODO()` to sidestep the canceled request context, that is the \"fresh background inside a request-scoped function\" finding: cancellation, deadlines, and tracing no longer propagate. The correct source is `r.Context()`, captured before the goroutine starts. (Conversely, if it *does* use `r.Context()` correctly, cancelation will likely abort its work mid-flight \u2014 another reason fire-and-forget is wrong here.)\n\n4. **Silent failure.** A fire-and-forget goroutine has nowhere to return an error. Any error inside it is dropped or panics the process (an unrecovered panic in a spawned goroutine crashes the whole server, unlike in a handler where `net/http` recovers). Flag a bare dropped error, and flag the panic exposure.\n\n5. **No shutdown hook.** The server's graceful shutdown (`Server.Shutdown` waiting on in-flight handlers) will not wait for this goroutine. Work is silently lost on shutdown.\n\nFix direction (not applied \u2014 this skill is read-only):\n\n- If the work must complete as part of the response, do it synchronously in the handler, or join it before returning via a `sync.WaitGroup` / `errgroup.Group` that the handler `Wait()`s on.\n- If it must be asynchronous, do not spawn it from the handler at all \u2014 hand it to a long-lived worker pool or a bounded queue with a real owner, propagation of `r.Context()`'s values where appropriate, error reporting, and lifecycle tied to the server rather than the request.\n- Whichever path, give the goroutine a join point or a stated, deliberate reason it is fire-and-forget, guard all shared state with a mutex/channel/`sync/atomic`, and use `r.Context()` (captured before the spawn) rather than `context.Background()`.\n\nA suspected race here is not provable by reading alone, so the review would add: run `go test -race ./...` to confirm.\n\nSince the diff was described rather than shown, I could not read the changed `.go` files, confirm whether the handler received a context it then discarded, or tell whether this pattern is newly introduced or pre-existing in the touched file. The finding above holds for the described shape; a full review would pin the exact file:line and classify it as diff-introduced versus pre-existing.",
807
+ "outputSha256": "05eaa878e97be9ac597f629add2dfe11d2ceb50b60beb6299953707bd4614a4c",
808
+ "promptSha256": "fb0523ac155676acf94fab3562929cc0ace8913061ff9015aa0dd3b3337117fd",
809
+ "deterministic": [],
810
+ "judge": {
811
+ "verdict": "pass",
812
+ "reason": "Answer explicitly names the missing join point ('no WaitGroup, errgroup, or channel receive'), explains the leak consequence ('goroutine leaks past the handler's return; it outlives the request'), and names fix directions (sync.WaitGroup/errgroup.Group join before returning) as a finding, explicitly 'not applied'. No hedge dismissing the missing join as acceptable."
813
+ },
814
+ "passed": true
815
+ },
816
+ {
817
+ "output": "A goroutine started inside a request handler with no join point is a goroutine leak and a lifecycle bug, and it's worse than a plain leak because the handler's request scope normally owns cancellation.\n\n## The finding\n\n```\nhandler.go:_ \u2014 goroutine started in the request handler has no join point\n (no WaitGroup, no channel receive, no errgroup, ctx not checked).\n Risk: outlives the request; leak + race with handler return.\n```\n\n## Why it's wrong\n\n**No join point means no ownership.** Nothing in the diff waits for, tracks, or reports on the goroutine. The handler returns as soon as it schedules the work, so:\n\n- **It leaks if it never finishes.** Each inbound request spawns one goroutine. If the work blocks (a downstream RPC, a lock, an unbuffered send nobody receives), that goroutine lives forever. Under sustained traffic this is an unbounded goroutine + memory leak that only shows up under load \u2014 the classic shape.\n- **It's a data race waiting to happen.** The handler almost certainly closes over locals or its response state. The handler returns, those locals are still being touched by the goroutine. If the goroutine writes anything the caller also reads (a shared map, a captured struct field, a `*http.ResponseWriter`), that's an unsynchronized concurrent access \u2014 `go test -race ./...` is the way to confirm, don't assert the race from reading alone.\n- **Cancellation is severed.** This is the context critique. The handler has `r.Context()`, which the server cancels when the client disconnects or the request completes. A floating goroutine either doesn't receive that context at all, or (worse) receives it and ignores it. Either way, client cancellation no longer stops the work \u2014 you're doing full-cost work for a client who already left, and holding resources the server expected to reclaim at end-of-request.\n- **Errors vanish.** With no join point there's no `errgroup`, no error channel, nothing to collect the failure. The goroutine's error has nowhere to go, so failures become silent \u2014 exactly the ignored-error category this review checks.\n\n**If the intent was fire-and-forget**, that intent needs to be stated explicitly in the code and it still isn't free: detached work must be tied to a lifetime that outlives the request deliberately (e.g. a root context the server owns), still needs a bound on how many such goroutines can exist, and still needs the error surfaced somewhere. \"No join point\" and \"deliberately detached\" are not the same claim, and the diff currently makes neither.\n\n## Fix direction (not applied \u2014 this review is read-only)\n\n- If the caller should wait: wrap it in an `errgroup.Group` (or pass a `WaitGroup` the handler `Wait()`s on) and `Wait()` before returning; propagate its error.\n- If it must outlive the request: give it a lifetime that isn't the request \u2014 a server-scoped context \u2014 bound the concurrency (a worker pool or semaphore), and route errors to a logger/metrics sink so they don't disappear.\n- Either way, check the context inside the work loop and select on `ctx.Done()` so cancellation actually propagates.\n\nOne caveat to verify while you're in there: confirm the goroutine isn't also writing a captured map or shared field with no lock \u2014 that's a separate Step 2 finding (nil-map write / unsynchronized map access) on top of the leak.",
818
+ "outputSha256": "acee6a3c2e5f2a01c54d4802f4912f7e4c5831a349053913ac1874dec776a0d8",
819
+ "promptSha256": "fb0523ac155676acf94fab3562929cc0ace8913061ff9015aa0dd3b3337117fd",
820
+ "deterministic": [],
821
+ "judge": {
822
+ "verdict": "pass",
823
+ "reason": "Criterion 1: answer states 'goroutine started in the request handler has no join point (no WaitGroup, no channel receive, no errgroup)'. Criterion 2: explains it 'outlives the request' and 'leaks if it never finishes', cancellation severed. Criterion 3: names errgroup.Group, WaitGroup, channel as fix direction, explicitly 'not applied \u2014 this review is read-only'. No fail criterion: it does not dismiss the missing join as acceptable."
824
+ },
825
+ "passed": true
826
+ },
827
+ {
828
+ "output": "The problem: that goroutine has no join point, so nothing in the handler can wait for it to finish, and nothing bounds its lifetime to the request.\n\nConcretely, what's wrong:\n\n**1. It outlives the request.** A request handler returns as soon as it writes its response. A goroutine started inside it with no `WaitGroup`/`errgroup`/channel receive keeps running after the handler returns. There is no point in the code that says \"this handler is not done until this goroutine is done.\" That is a goroutine leak by construction \u2014 every request that takes this path leaves a goroutine behind for an unbounded time.\n\n**2. It likely captures request-scoped state.** If the goroutine closes over `r.Context()` or anything derived from it, the context is cancelled when the handler returns, so the goroutine's own work (DB call, outbound HTTP, whatever) fails immediately or intermittently \u2014 a race between \"handler returns\" and \"goroutine gets to use the context.\" If instead it uses `context.Background()`/`context.TODO()`, it's the opposite failure: it has no cancellation at all, so a slow downstream call can hang forever and never observe shutdown. Both are context misuse; a goroutine that must outlive the request needs a context whose lifetime is owned by something other than the request (a server-scoped context, a job queue's context), not `Background()` chosen by accident.\n\n**3. Nothing propagates its error.** A fire-and-forget goroutine has no one to return an error to. If it fails, the failure is silent unless it explicitly logs \u2014 and even then, the client already got a 200. If the work is meant to be best-effort, that needs to be a stated, deliberate decision, not the default outcome of forgetting a join.\n\n**4. It's outside the concurrency budget.** No `WaitGroup` also usually means no limit on how many such goroutines can be in flight. Under load, one request-per-goroutine with no bound is an unbounded fan-out \u2014 a small traffic spike turns into unbounded goroutines and unbounded memory.\n\n**5. Shutdown races.** On server shutdown, the handler returns, `http.Server.Shutdown` completes, and the process exits \u2014 while these goroutines are mid-write to a DB or file. That's a truncated write or a lost update, and it's nondeterministic, so it typically shows up only in production.\n\nFix direction (not applied \u2014 this is a read-only review):\n- If the work must complete before the response, do it inline, or block on it with an `errgroup.Group`/`WaitGroup` before writing the response.\n- If it's genuinely meant to be async, hand it to something with an owned lifetime and a bounded pool \u2014 a worker queue or `errgroup` with a concurrency limit \u2014 and give it a context owned by that subsystem, not the request's context and not a bare `context.Background()`.\n- Make the failure path explicit: log the error and record the outcome somewhere observable, since no caller can receive it.\n- Because this is a race/lifetime question, confirm with `go test -race ./...` and a load test under `-race` rather than concluding from reading alone.\n\nOne more thing worth checking before reporting: whether this goroutine is new in the diff or pre-existing in code the diff only touches. If it's pre-existing, note it separately from anything the diff introduces.",
829
+ "outputSha256": "dec2fb13249e6c24ef3f3f066dd6f56ae200f686fd7940a1b6f7053c1486a150",
830
+ "promptSha256": "fb0523ac155676acf94fab3562929cc0ace8913061ff9015aa0dd3b3337117fd",
831
+ "deterministic": [],
832
+ "judge": {
833
+ "verdict": "pass",
834
+ "reason": "Answer explicitly states the goroutine 'has no join point' and 'no WaitGroup/errgroup/channel receive' (crit.1). It explains concrete consequences: 'keeps running after the handler returns', 'goroutine leak by construction', and shutdown races where process exits 'while these goroutines are mid-write' (crit.2). It names fix directions as findings, not applied: 'block on it with an errgroup.Group/WaitGroup', worker queue with bounded pool (crit.3). No hedge dismissing the missing join as acceptable."
835
+ },
836
+ "passed": true
837
+ },
838
+ {
839
+ "output": "The finding lands at the intersection of Step 2's \"Goroutines and concurrency\" check and its \"Context\" check, so it's worth reporting both angles.\n\n**The core defect: no join point.** A goroutine started inside a request handler with no `WaitGroup`, no `errgroup`, and no channel to receive from is fire-and-forget. Per Step 2, that's only acceptable when it's *deliberately* fire-and-forget with a stated reason. If the diff doesn't state one, flag it. The concrete risks:\n\n- **The handler returns before the work finishes.** The response goes out (or the connection closes) while the goroutine is still running. Anything it was supposed to accomplish \u2014 writing a result, sending a notification, flushing a log \u2014 races the handler's return and may never be observed.\n- **Leak under cancellation.** The request is canceled or times out, the handler returns, but the goroutine has nothing telling it to stop, so it runs to completion anyway. With a loop or a blocking call inside, it can outlive the request indefinitely.\n- **No error propagation.** The `error` the goroutine produces has no receiver. It's either silently dropped (a \"bare dropped error\" per the Errors check) or logged somewhere the caller never sees, so the operation fails invisibly.\n\nReport shape, per Step 3:\n\n```\ninternal/api/handler.go:NN \u2014 goroutine started in ServeHTTP has no join point\n (no WaitGroup/errgroup, no result channel, ctx not checked). Risk: leaks\n past the handler's return under caller cancellation; the error it produces\n has no receiver and is silently dropped. Fix direction: either join it\n (errgroup.Group the handler waits on before writing the response) or make\n the fire-and-forget intent explicit and pass r.Context() so cancellation\n reaches it.\n```\n\n**The second angle: context.** Check what context the goroutine uses, because the two defects usually co-occur. If it calls `context.Background()` or `context.TODO()` instead of `r.Context()`, that's the Step 2 context-misuse flag \u2014 the handler *received* a request context and discarded it. With `r.Context()` the goroutine at least learns about cancellation and deadline; with `Background()` it never does. Note that passing `r.Context()` fixes cancellation propagation but does **not** by itself provide a join point \u2014 those are separate findings and both need reporting.\n\nA couple of things to pin down before asserting, per the Rules:\n\n- **Is it deliberate?** If the diff has a comment like `// fire-and-forget: audit log write, best-effort` and the goroutine is genuinely self-contained, downgrade to a note rather than a defect \u2014 but a handler is a request-scoped path, so the burden is on the code to say why the request may return first.\n- **Diff-introduced or pre-existing?** If this handler predates the change and the diff merely touched a nearby line, say so and separate it from findings the diff actually introduces (Step 1 / Rules).\n- **Don't over-claim a race.** Whether the goroutine races with anything the handler reads depends on the surrounding code. If I can't see that from the diff alone, I'll say \"run `go test -race ./...` to confirm\" rather than assert a data race exists.\n\nI'm reporting only \u2014 no edits, per the read-only rule.",
840
+ "outputSha256": "c0469dc2274c47e1ecf049176a58aa85a772f0166b193fcd685f50ff647ff2d2",
841
+ "promptSha256": "fb0523ac155676acf94fab3562929cc0ace8913061ff9015aa0dd3b3337117fd",
842
+ "deterministic": [],
843
+ "judge": {
844
+ "verdict": "pass",
845
+ "reason": "Answer explicitly identifies the missing join point ('no WaitGroup, no errgroup, and no channel to receive from is fire-and-forget'), gives concrete consequences (handler returns before work finishes; goroutine leaks past handler return under cancellation), and names fix directions (errgroup.Group the handler waits on, or channel) as findings, not applied edits. No hedge dismissing the missing join as acceptable."
846
+ },
847
+ "passed": true
848
+ }
849
+ ]
850
+ }
851
+ ],
852
+ "verdict": "pass",
853
+ "scope": "bundled",
854
+ "skillDigest": "5cd84a4cb21150ea88a49e1696108695d435251592492a0016a75910078abf51",
855
+ "catalogDigest": "4f4016d410837e394a27e5b247e38ef2f57a1ee0baba4436ce7d3d71e223333d",
856
+ "judgePromptVersion": "2026-09-25.1",
857
+ "runner": "deepseek",
858
+ "model": "deepseek-chat",
859
+ "runnerPromptVersion": "2026-09-25.1",
860
+ "recordedAt": "2026-09-25T05:07:45.255Z",
861
+ "judge": "deepseek",
862
+ "judgeModel": "deepseek-chat"
863
+ },
864
+ {
865
+ "schemaVersion": "1.0.0",
866
+ "skillId": "go/go-implementation",
867
+ "strictness": "high",
868
+ "trials": 10,
869
+ "triggerAccuracy": {
870
+ "truePositive": 6,
871
+ "falsePositive": 0,
872
+ "positives": 6,
873
+ "negatives": 6
874
+ },
875
+ "evidence": "authored",
876
+ "scenarios": [
877
+ {
878
+ "id": "trigger-positive-1",
879
+ "kind": "trigger-positive",
880
+ "prompt": "Implement a new Go package under internal/order that wraps errors with %w",
881
+ "strictness": "high",
882
+ "trials": 1,
883
+ "passes": 1,
884
+ "passRate": 1,
885
+ "passAtK": 1,
886
+ "grader": "trigger-rank-fork-family",
887
+ "status": "ran",
888
+ "deterministic": true
889
+ },
890
+ {
891
+ "id": "trigger-positive-2",
892
+ "kind": "trigger-positive",
893
+ "prompt": "Add a Go worker pool under cmd/worker that bounds concurrency with errgroup",
894
+ "strictness": "high",
895
+ "trials": 1,
896
+ "passes": 1,
897
+ "passRate": 1,
898
+ "passAtK": 1,
899
+ "grader": "trigger-rank-fork-family",
900
+ "status": "ran",
901
+ "deterministic": true
902
+ },
903
+ {
904
+ "id": "trigger-positive-3",
905
+ "kind": "trigger-positive",
906
+ "prompt": "Add a Go endpoint and wire it into cmd/ and internal/, propagating context.Context correctly",
907
+ "strictness": "high",
908
+ "trials": 1,
909
+ "passes": 1,
910
+ "passRate": 1,
911
+ "passAtK": 1,
912
+ "grader": "trigger-rank-fork-family",
913
+ "status": "ran",
914
+ "deterministic": true
915
+ },
916
+ {
917
+ "id": "trigger-positive-4",
918
+ "kind": "trigger-positive",
919
+ "prompt": "What's the right Go interface placement for this new client -- producer or consumer package?",
920
+ "strictness": "high",
921
+ "trials": 1,
922
+ "passes": 1,
923
+ "passRate": 1,
924
+ "passAtK": 1,
925
+ "grader": "trigger-rank-fork-family",
926
+ "status": "ran",
927
+ "deterministic": true
928
+ },
929
+ {
930
+ "id": "trigger-positive-5",
931
+ "kind": "trigger-positive",
932
+ "prompt": "Add a goroutine in Go that watches a channel and joins with sync.WaitGroup",
933
+ "strictness": "high",
934
+ "trials": 1,
935
+ "passes": 1,
936
+ "passRate": 1,
937
+ "passAtK": 1,
938
+ "grader": "trigger-rank-fork-family",
939
+ "status": "ran",
940
+ "deterministic": true
941
+ },
942
+ {
943
+ "id": "trigger-positive-6",
944
+ "kind": "trigger-positive",
945
+ "prompt": "Implement this feature in a Go cmd/api binary using log/slog for structured logging",
946
+ "strictness": "high",
947
+ "trials": 1,
948
+ "passes": 1,
949
+ "passRate": 1,
950
+ "passAtK": 1,
951
+ "grader": "trigger-rank-fork-family",
952
+ "status": "ran",
953
+ "deterministic": true
954
+ },
955
+ {
956
+ "id": "trigger-negative-1",
957
+ "kind": "trigger-negative",
958
+ "prompt": "Implement this feature in Rust using tokio for async",
959
+ "strictness": "high",
960
+ "trials": 1,
961
+ "passes": 1,
962
+ "passRate": 1,
963
+ "passAtK": 1,
964
+ "grader": "trigger-rank-fork-family",
965
+ "status": "ran",
966
+ "deterministic": true
967
+ },
968
+ {
969
+ "id": "trigger-negative-2",
970
+ "kind": "trigger-negative",
971
+ "prompt": "Add this endpoint in a Python FastAPI service",
972
+ "strictness": "high",
973
+ "trials": 1,
974
+ "passes": 1,
975
+ "passRate": 1,
976
+ "passAtK": 1,
977
+ "grader": "trigger-rank-fork-family",
978
+ "status": "ran",
979
+ "deterministic": true
980
+ },
981
+ {
982
+ "id": "trigger-negative-3",
983
+ "kind": "trigger-negative",
984
+ "prompt": "Implement this React component with the new form fields",
985
+ "strictness": "high",
986
+ "trials": 1,
987
+ "passes": 1,
988
+ "passRate": 1,
989
+ "passAtK": 1,
990
+ "grader": "trigger-rank-fork-family",
991
+ "status": "ran",
992
+ "deterministic": true
993
+ },
994
+ {
995
+ "id": "trigger-negative-4",
996
+ "kind": "trigger-negative",
997
+ "prompt": "Review this Go diff for goroutine leaks and data races",
998
+ "strictness": "high",
999
+ "trials": 1,
1000
+ "passes": 1,
1001
+ "passRate": 1,
1002
+ "passAtK": 1,
1003
+ "grader": "trigger-rank-fork-family",
1004
+ "status": "ran",
1005
+ "deterministic": true
1006
+ },
1007
+ {
1008
+ "id": "trigger-negative-5",
1009
+ "kind": "trigger-negative",
1010
+ "prompt": "Fix this failing go test -race in the worker package",
1011
+ "strictness": "high",
1012
+ "trials": 1,
1013
+ "passes": 1,
1014
+ "passRate": 1,
1015
+ "passAtK": 1,
1016
+ "grader": "trigger-rank-fork-family",
1017
+ "status": "ran",
1018
+ "deterministic": true
1019
+ },
1020
+ {
1021
+ "id": "trigger-negative-6",
1022
+ "kind": "trigger-negative",
1023
+ "prompt": "Write table-driven pytest cases for this Python function",
1024
+ "strictness": "high",
1025
+ "trials": 1,
1026
+ "passes": 1,
1027
+ "passRate": 1,
1028
+ "passAtK": 1,
1029
+ "grader": "trigger-rank-fork-family",
1030
+ "status": "ran",
1031
+ "deterministic": true
1032
+ },
1033
+ {
1034
+ "id": "context-propagation",
1035
+ "kind": "behavior",
1036
+ "prompt": "I'm adding a new function to internal/order that calls a downstream client. How should I handle the context and errors?",
1037
+ "strictness": "high",
1038
+ "trials": 10,
1039
+ "passes": 10,
1040
+ "passRate": 1,
1041
+ "passAtK": 1,
1042
+ "grader": "regex+judge",
1043
+ "status": "ran",
1044
+ "trialRecords": [
1045
+ {
1046
+ "output": "For a function in `internal/order` calling a downstream client, follow these rules.\n\n**Context**\n\n- Accept `ctx context.Context` as the first parameter \u2014 no exceptions, even if this function doesn't use it directly, because the client call below it needs it:\n ```go\n func (s *Service) FetchOrder(ctx context.Context, id string) (Order, error) {\n o, err := s.client.GetOrder(ctx, id)\n ...\n }\n ```\n- Forward that same `ctx` down to the client. Do not create a fresh `context.Background()` inside the chain \u2014 that silently severs cancellation, deadlines, and values from the caller.\n- If you need a tighter deadline than the caller's for this specific downstream call, derive it with `context.WithTimeout(ctx, d)` and always `defer cancel()`:\n ```go\n ctx, cancel := context.WithTimeout(ctx, downstreamTimeout)\n defer cancel()\n ```\n- Never store the `ctx` on the `Service` struct field. Thread it through.\n\n**Errors**\n\n- Wrap with `%w` at each layer boundary so callers can unwrap:\n ```go\n o, err := s.client.GetOrder(ctx, id)\n if err != nil {\n return Order{}, fmt.Errorf(\"fetching order %s: %w\", id, err)\n }\n ```\n- Reach for `%v` only when you deliberately want to cut the chain \u2014 rare; the caller you can't see yet is who `errors.Is`/`errors.As` serves, so keep the chain open by default.\n- If your package defines sentinel errors (e.g. `ErrOrderNotFound`), return them directly or wrapped, and let downstream callers test with `errors.Is`. Don't compare error strings.\n- Check every error. If you truly must discard one, make it explicit and commented (`_ = resp.Body.Close() // best-effort`), never a silent `if err != nil {}`.\n\nA typical shape:\n\n```go\nfunc (s *Service) FetchOrder(ctx context.Context, id string) (Order, error) {\n ctx, cancel := context.WithTimeout(ctx, s.timeout)\n defer cancel()\n\n o, err := s.client.GetOrder(ctx, id)\n if err != nil {\n if errors.Is(err, client.ErrNotFound) {\n return Order{}, fmt.Errorf(\"fetching order %s: %w\", id, ErrOrderNotFound)\n }\n return Order{}, fmt.Errorf(\"fetching order %s: %w\", id, err)\n }\n return o, nil\n}\n```\n\nBefore you finish: run `go build ./...`, `go vet ./...`, and `go test -race ./...`; confirm `gofmt -l` is clean. If you spawn any goroutine off this call, give it a join point (`errgroup`/`WaitGroup`) and the same `ctx` for cancellation.",
1047
+ "outputSha256": "d49bceb201cdfc0b7ad27df65a2c5789f91ac0a38b67f7d285edfadd30498624",
1048
+ "promptSha256": "9712a2b3973edb8bd2f0db6bd2f4cda7d89edf826fc0f006925bf1189b24e2c3",
1049
+ "deterministic": [
1050
+ true
1051
+ ],
1052
+ "judge": {
1053
+ "verdict": "pass",
1054
+ "reason": "Criterion 1: 'Accept ctx context.Context as the first parameter' and 'Never store the ctx on the Service struct field. Thread it through.' Criterion 2: concrete wrap shown: `return Order{}, fmt.Errorf(\"fetching order %s: %w\", id, err)`. Criterion 3: 'so callers can unwrap' and 'errors.Is/errors.As serves'. No fail criteria: discarding only warned against with comment."
1055
+ },
1056
+ "passed": true
1057
+ },
1058
+ {
1059
+ "output": "Take the context as the first parameter and thread it through unchanged \u2014 do not store it on a struct, do not replace it with `context.Background()` inside the call chain, and do not add a new deadline the caller didn't ask for.\n\n```go\n// internal/order/service.go\npackage order\n\nimport (\n\t\"context\"\n\t\"fmt\"\n)\n\n// OrderClient is defined here, at the consumer, sized to what OrderService\n// actually calls -- not next to whatever concrete type implements it.\ntype OrderClient interface {\n\tSend(ctx context.Context, o Order) error\n}\n\ntype Service struct {\n\tclient OrderClient\n}\n\nfunc NewService(client OrderClient) *Service {\n\treturn &Service{client: client}\n}\n\nfunc (s *Service) Submit(ctx context.Context, o Order) error {\n\t// ... validation / mapping on the caller's ctx ...\n\n\tif err := s.client.Send(ctx, o); err != nil {\n\t\treturn fmt.Errorf(\"submit order %s: %w\", o.ID, err)\n\t}\n\treturn nil\n}\n```\n\nContext:\n\n- First parameter, always. The caller's `ctx` carries their cancellation, deadline, and values; substituting a fresh one anywhere on the path silently breaks all three for everyone upstream.\n- Only derive (`context.WithTimeout`, `WithCancel`) if this function genuinely owns a tighter bound than the caller set \u2014 and if you do, `defer cancel()` immediately. Don't stack a timeout \"just to be safe.\"\n- No context on struct fields. If you're tempted, it's a sign the method should take it as a param.\n\nErrors:\n\n- Wrap at each meaningful layer boundary with context about what was being attempted: `fmt.Errorf(\"submit order %s: %w\", o.ID, err)`. `%w` keeps the chain unwrappable; `%v` there is a mistake even if no current caller unwraps it.\n- Never swallow: no bare `_ = err`, no empty `if err != nil {}`. Check it, wrap it, or return it. If you truly must discard, comment why and route it to a log with `slog` rather than dropping it silently.\n- Callers compare with `errors.Is`/`errors.As`, never string matching \u2014 which is exactly why the wrap verbs matter. Make sure the sentinel/typed errors the downstream client returns stay reachable through your wraps.\n\nTwo things to confirm against the project first:\n\n1. Check `go.mod`'s `go` directive and mirror the error/logging style of the neighboring files in `internal/order` \u2014 if they already use `log/slog` with structured fields, match it; if they use a different logger, don't introduce `slog` for one function.\n2. If this call can fan out or run concurrently with others, the goroutines go under an `errgroup.Group` bound to the same `ctx`, so the first error cancels the rest and you join before returning \u2014 no goroutine outlives `Submit`.\n\nVerification before you call it done: `go build ./...`, `go vet ./...`, `go test -race ./...` all green, and `gofmt -l` clean.",
1060
+ "outputSha256": "ca916959314a10f5cc9b01bf03aa9d135fee3a55e53eab04dbef5f1b9ad86270",
1061
+ "promptSha256": "9712a2b3973edb8bd2f0db6bd2f4cda7d89edf826fc0f006925bf1189b24e2c3",
1062
+ "deterministic": [
1063
+ true
1064
+ ],
1065
+ "judge": {
1066
+ "verdict": "pass",
1067
+ "reason": "Criterion 1: 'Take the context as the first parameter and thread it through unchanged \u2014 do not store it on a struct' plus Submit(ctx context.Context,...) passing ctx to s.client.Send(ctx, o). Criterion 2: concrete wrap `return fmt.Errorf(\"submit order %s: %w\", o.ID, err)` returned to caller. Criterion 3: 'Callers compare with errors.Is/errors.As ... which is exactly why the wrap verbs matter.' No fail criteria met."
1068
+ },
1069
+ "passed": true
1070
+ },
1071
+ {
1072
+ "output": "For a downstream call inside `internal/order`, thread context as a parameter and wrap errors at the boundary:\n\n```go\nfunc (s *Service) LookupOrder(ctx context.Context, id string) (*Order, error) {\n o, err := s.client.FetchOrder(ctx, id)\n if err != nil {\n return nil, fmt.Errorf(\"lookup order %s: %w\", id, err)\n }\n return o, nil\n}\n```\n\nKey points specific to this shape:\n\n**Context**\n- `ctx` is the first parameter, `context.Context` \u2014 never a struct field on `Service`. The caller (likely `cmd/api`'s handler via `r.Context()`) owns the deadline and cancellation; you only forward.\n- Pass the *same* `ctx` down to `s.client.FetchOrder`. Do not call `context.Background()` inside the service to \"decouple\" \u2014 that silently breaks cancellation and deadline propagation from the handler.\n- If this downstream call can be slow and you want a tighter bound than the caller's, derive with `context.WithTimeout(ctx, ...)` and `defer cancel()` \u2014 but derive *from* the incoming ctx, don't replace it.\n\n**Errors**\n- Wrap once, at this function's boundary, with `%w` and enough context to identify the operation and the input (`\"lookup order %s\"`). Do not wrap at every internal hop \u2014 one meaningful wrap per layer boundary.\n- Return `nil` for the value on error, don't return a partially-populated `*Order`.\n- Let callers use `errors.Is`/`errors.As` against whatever the client returns; because you used `%w`, the chain is preserved. If the client returns a sentinel (e.g. `order.ErrNotFound`), a caller can still do `errors.Is(err, order.ErrNotFound)` through your wrap.\n- Don't translate errors into a generic `errors.New(\"failed to lookup order\")` \u2014 that destroys the unwrap chain.\n\n**Concurrency, if you fan out**\nIf this function ends up making several downstream calls, use `errgroup.Group` (from `golang.org/x/sync/errgroup` \u2014 check `go.mod` first) with `eg, ctx := errgroup.WithContext(ctx)` and `eg.Go(func() error {...})`, then `return eg.Wait()`. That gives you first-error return plus shared cancellation; each goroutine reads the derived `ctx`, so one failure cancels the rest.\n\nBefore finishing, run `go build ./...`, `go vet ./...`, and `go test -race ./...` \u2014 the `-race` flag matters here because a downstream client call plus any goroutines is exactly where races surface. If a test fails, fix the logic; this isn't a build-plumbing problem.\n\nOne check against your project: confirm the `go` directive in `go.mod` before reaching for newer stdlib idioms (e.g. `WaitGroup.Go` needs Go 1.25), and match whether the package already logs via `log/slog` or something else \u2014 use whatever is already standardized there.",
1073
+ "outputSha256": "ce99ea09fabd5f82bf25bd3b267e17c05441a28d544b6cd8403bc156da7be77d",
1074
+ "promptSha256": "9712a2b3973edb8bd2f0db6bd2f4cda7d89edf826fc0f006925bf1189b24e2c3",
1075
+ "deterministic": [
1076
+ true
1077
+ ],
1078
+ "judge": {
1079
+ "verdict": "pass",
1080
+ "reason": "Criterion 1: 'ctx is the first parameter, context.Context \u2014 never a struct field on Service' and 'Pass the same ctx down to s.client.FetchOrder'. Criterion 2: concrete code `return nil, fmt.Errorf(\"lookup order %s: %w\", id, err)` returns wrapped error. Criterion 3: 'Let callers use errors.Is/errors.As ... because you used %w, the chain is preserved'. No fail criteria met."
1081
+ },
1082
+ "passed": true
1083
+ },
1084
+ {
1085
+ "output": "For both, the rule is the same shape: the function you're adding should be transparent to `context.Context` and opaque-respecting to errors.\n\n## Context\n\n`context.Context` goes in as the first parameter, named `ctx`, and gets forwarded unchanged to the downstream client call:\n\n```go\nfunc (s *OrderService) PlaceOrder(ctx context.Context, req PlaceOrderRequest) (Order, error) {\n // ... validation ...\n resp, err := s.client.Submit(ctx, req)\n if err != nil {\n return Order{}, fmt.Errorf(\"placing order %q: %w\", req.OrderID, err)\n }\n // ...\n}\n```\n\nThings that matter here:\n\n- **Do not** store `ctx` on `OrderService` or any struct field. If the current method doesn't take a `ctx`, add one \u2014 an internal function that calls a downstream client always needs the caller's cancellation and deadline.\n- **Do not** pass `context.Background()` or `context.TODO()` at the downstream call site. If the caller didn't supply a deadline, that's the caller's call to make; inserting a fresh background context silently breaks timeout propagation from the HTTP handler or CLI that invoked you.\n- If you need to bound the downstream call, derive it from the incoming `ctx` (`ctx, cancel := context.WithTimeout(ctx, s.clientTimeout); defer cancel()`), never from a fresh root.\n- If you spawn any goroutine inside this function, it must select on `ctx.Done()` and have a join point \u2014 see the goroutine rules in the skill. A downstream call that itself takes `ctx` handles that for you; a hand-rolled worker does not.\n\n## Errors\n\nWrap every error that crosses a meaningful boundary with `fmt.Errorf` and `%w`, naming the operation in the message:\n\n```go\nresp, err := s.client.Submit(ctx, req)\nif err != nil {\n return Order{}, fmt.Errorf(\"submitting order to payment client: %w\", err)\n}\n```\n\n- Use `%w`, not `%v`. `%w` keeps the chain unwrappable so callers can do `errors.Is(err, downstream.ErrRateLimited)` or `errors.As(err, &apiErr)`. It costs nothing.\n- One wrap per layer is plenty. If the downstream client already wrapped with its own context (`\"submitting order: dial tcp ...\"`), you don't need to restate the whole thing \u2014 wrap with *your* layer's context (\"what was I doing when this failed\"), not a paraphrase of the inner message.\n- Check every error. No empty `if err != nil {}`, no naked `_ = err` without a comment explaining why it's safe to drop.\n- If the function needs to distinguish failure modes for its caller (e.g. a not-found vs. a transport failure), return a sentinel via `errors.Is` or a typed error and document it; don't make callers match on strings.\n- Don't reach for `panic` in a library/`internal` function for an error the caller could handle.\n\n## Putting it together\n\nA worked skeleton for `internal/order`:\n\n```go\nfunc (s *OrderService) PlaceOrder(ctx context.Context, req PlaceOrderRequest) (Order, error) {\n if err := req.Validate(); err != nil {\n return Order{}, fmt.Errorf(\"validating order request: %w\", err)\n }\n\n ctx, cancel := context.WithTimeout(ctx, s.downstreamTimeout)\n defer cancel()\n\n resp, err := s.client.Submit(ctx, req)\n if err != nil {\n return Order{}, fmt.Errorf(\"submitting order %q to client: %w\", req.OrderID, err)\n }\n if resp.Status != ClientStatusAccepted {\n return Order{}, fmt.Errorf(\"order %q rejected by client: %w\", req.OrderID, ErrOrderRejected)\n }\n\n return toOrder(resp), nil\n}\n```\n\nThat gives you: caller's deadline honored, downstream call bounded, both failure paths wrapped with `%w` so a caller can `errors.Is(err, ErrOrderRejected)` and still see the underlying transport error via `errors.As`.\n\nBefore you report done: `go build ./...`, `go vet ./...`, `go test -race ./...` all clean, and every new error return is checked or wrapped.",
1086
+ "outputSha256": "c8bc43f0a123073b6c1101332b604d60c012a1921b9c82958051b53a507720f3",
1087
+ "promptSha256": "9712a2b3973edb8bd2f0db6bd2f4cda7d89edf826fc0f006925bf1189b24e2c3",
1088
+ "deterministic": [
1089
+ true
1090
+ ],
1091
+ "judge": {
1092
+ "verdict": "pass",
1093
+ "reason": "Criterion 1: states ctx as first param forwarded unchanged to s.client.Submit(ctx, req) and explicitly 'Do not store ctx on OrderService or any struct field.' Criterion 2: concrete wrap `fmt.Errorf(\"submitting order %q to client: %w\", req.OrderID, err)` returned to caller. Criterion 3: connects wrap to errors.Is(err, ErrOrderRejected)/errors.As. No fail criteria met."
1094
+ },
1095
+ "passed": true
1096
+ },
1097
+ {
1098
+ "output": "For a new function in `internal/order` that calls a downstream client:\n\n**Context**\n\n- Take `ctx context.Context` as the first parameter: `func (s *Service) PlaceOrder(ctx context.Context, req PlaceOrderRequest) (Order, error)`.\n- Forward that same `ctx` to every downstream call \u2014 the client's own method signature should already accept a context, so pass it straight through. Never substitute `context.Background()` or `context.TODO()` inside the call chain; that severs the caller's cancellation and deadline.\n- Do not stash the context on the `Service` struct field. If `Service` is constructed once at startup, it has no per-request context; the context only exists per call.\n- If you spawn goroutines here (e.g. fan out to the client and a validator concurrently), use `errgroup.WithContext(ctx)` so the first error cancels siblings, and always `g.Wait()` before returning \u2014 that's your join point.\n- If the downstream call needs a tighter bound than the caller's, derive it with `ctx, cancel := context.WithTimeout(ctx, d)` and `defer cancel()`. Only tighten; never loosen.\n\n**Errors**\n\n- Wrap at this layer with context about what was being attempted:\n ```go\n order, err := s.client.Create(ctx, req)\n if err != nil {\n return Order{}, fmt.Errorf(\"placing order %s: %w\", req.ID, err)\n }\n ```\n Use `%w`, not `%v`, so callers upstream can `errors.Is`/`errors.As` against the downstream client's sentinel or typed errors.\n- Don't double-wrap. If `s.client.Create` already returns `fmt.Errorf(\"create order: %w\", err)`, adding another `\"placing order: %w\"` is fine only if it names a genuinely different layer/operation; otherwise just `return err`.\n- Inspect downstream errors with `errors.Is`/`errors.As` rather than string matching, e.g. to translate a client's `ErrNotFound` into your domain's not-found, or to decide retryability.\n- Never swallow: no bare `if err != nil {}`, no `_ = err`. Either return it (wrapped), handle it explicitly, or discard it with a comment explaining why it's safe.\n- For a partial failure where you must still return a value, wrap the error and return it alongside whatever state the caller can meaningfully use \u2014 don't invent success.\n\n**Verify**: `go build ./...`, `go vet ./...`, `go test -race ./...` all green; `gofmt -l` clean. Watch for the four red flags \u2014 interface/goroutine placement is fine here, but `context.Background()` substitution and `%v`-instead-of-`%w` are the easy mistakes in a downstream-client wrapper.",
1099
+ "outputSha256": "c4d9ddeb1d1709e195a03ead1eee96c2bbb4de33f0d8f9d3bee574bfa86db416",
1100
+ "promptSha256": "9712a2b3973edb8bd2f0db6bd2f4cda7d89edf826fc0f006925bf1189b24e2c3",
1101
+ "deterministic": [
1102
+ true
1103
+ ],
1104
+ "judge": {
1105
+ "verdict": "pass",
1106
+ "reason": "Criterion 1: answer states ctx as first parameter `func (s *Service) PlaceOrder(ctx context.Context, ...)` and says 'Do not stash the context on the Service struct field.' Criterion 2: concrete fix shown: `return Order{}, fmt.Errorf(\"placing order %s: %w\", req.ID, err)` with %w. Criterion 3: connects wrap to callers using errors.Is/errors.As. No fail criteria: explicitly warns against `_ = err` and %v."
1107
+ },
1108
+ "passed": true
1109
+ },
1110
+ {
1111
+ "output": "# Context and errors for a downstream client call in `internal/order`\n\n## Context\n\nAccept `ctx context.Context` as the **first parameter** of the new function \u2014 don't reach for `context.Background()` or `context.TODO()` inside it:\n\n```go\nfunc (s *Service) GetOrder(ctx context.Context, id string) (*Order, error) {\n o, err := s.client.FetchOrder(ctx, id)\n ...\n}\n```\n\nThread it straight through to the downstream client call \u2014 the client is the thing that actually uses the cancellation/deadline/values, so it needs the same `ctx` you received. Do not store it on a struct field; pass it per-call so each request gets its own cancellation scope and values.\n\nIf your caller needs a tighter budget than theirs, derive it at the boundary where the work starts, not in the middle:\n\n```go\nctx, cancel := context.WithTimeout(ctx, 2*time.Second)\ndefer cancel()\n```\n\nTwo things to watch for:\n\n- Derive the child `ctx` **once** at the top of the function that owns the deadline, not at every layer below it \u2014 repeated `WithTimeout` wrapping compounds and obscures which deadline actually fires.\n- If the downstream client call itself starts goroutines (retries, hedged requests), it should own joining/cancelling them; don't fan out inside `GetOrder` without a `WaitGroup`/`errgroup` to join on.\n\n## Errors\n\nWrap at the layer boundary with `%w` so callers can `errors.Is`/`errors.As` through you, and add context about *what was being done*, not just *where*:\n\n```go\no, err := s.client.FetchOrder(ctx, id)\nif err != nil {\n return nil, fmt.Errorf(\"fetching order %s: %w\", id, err)\n}\n```\n\nConcrete caveats for a downstream call:\n\n- **Don't leak upstream error text into responses.** The wrapped chain is for logs and `errors.Is`; translating to a user-facing message is the handler's job, not this function's.\n- **Preserve sentinel errors from the client.** If the client exposes `client.ErrNotFound`, wrapping with `%w` keeps `errors.Is(err, client.ErrNotFound)` working at your layer \u2014 don't collapse it to a plain `fmt.Errorf(\"order lookup failed\")` with no `%w`, or the caller can no longer distinguish \"not found\" from \"downstream is down\".\n- **Check the error once, at the point it happens.** No `if err != nil { return nil, err }` passthrough without a wrap, and no swallowing (`_ = err`) without a comment saying why it's safe to ignore.\n- **Don't double-log.** Either this function logs the wrapped error with structured fields and returns it, or it returns it cleanly for a caller to log \u2014 not both, or you get two log lines per failure.\n\n```go\nif err != nil {\n slog.ErrorContext(ctx, \"order fetch failed\", \"order_id\", id, \"err\", err)\n return nil, fmt.Errorf(\"fetching order %s: %w\", id, err)\n}\n```\n\nPrefer logging at the boundary that knows the request identity (handler), and wrapping quietly in `internal/order`; that keeps the log line count per incident at one.\n\n## Sanity check before you ship\n\n- `gofmt -l` clean, `go vet ./...` clean, `go test -race ./...` green.\n- The new function's `ctx` parameter is forwarded to **every** downstream call it makes, not just some.\n- `errors.Is(err, client.Err...)` still works when the caller checks the value your function returns.",
1112
+ "outputSha256": "d5f2d0aa193e9dcdf2bee674c481f68899aa8e003f6ebf55a7f6de190bb973b2",
1113
+ "promptSha256": "9712a2b3973edb8bd2f0db6bd2f4cda7d89edf826fc0f006925bf1189b24e2c3",
1114
+ "deterministic": [
1115
+ true
1116
+ ],
1117
+ "judge": {
1118
+ "verdict": "pass",
1119
+ "reason": "Criterion 1: answer shows ctx as first param (GetOrder(ctx context.Context, id string)) threaded to s.client.FetchOrder(ctx, id) and explicitly says 'Do not store it on a struct field'. Criterion 2: concrete wrap shown: fmt.Errorf(\"fetching order %s: %w\", id, err) returned to caller. Criterion 3: connects wrap to errors.Is/errors.As and sentinel preservation."
1120
+ },
1121
+ "passed": true
1122
+ },
1123
+ {
1124
+ "output": "For a function in `internal/order` that calls a downstream client:\n\n**Context**\n\n- Accept `ctx context.Context` as the first parameter, not stored on a struct field:\n ```go\n func (s *Service) Fetch(ctx context.Context, id string) (*Order, error)\n ```\n- Forward that same `ctx` to every downstream call (`s.client.Get(ctx, id)`) \u2014 do not substitute `context.Background()`, even if it feels like \"internal\" code. That silently breaks the caller's cancellation, deadline, and values (trace IDs, auth), and it is the single most common context bug.\n- If you need a tighter bound than the caller's deadline for this specific call, derive it: `ctx, cancel := context.WithTimeout(ctx, 2*time.Second); defer cancel()`. Never widen the caller's deadline.\n- Only start a background context at the true entry point of the process (an HTTP handler's `r.Context()`, a CLI's `signal.NotifyContext`), never mid-chain.\n\n**Errors**\n\nWrap at the layer boundary so the message says *what this layer was doing*, and keep the chain intact with `%w`:\n```go\no, err := s.client.Get(ctx, id)\nif err != nil {\n return nil, fmt.Errorf(\"fetching order %s: %w\", id, err)\n}\n```\n- Use `%w`, not `%v` \u2014 the caller you can't see yet is exactly who needs `errors.Is`/`errors.As`. `errors.Is(err, order.ErrNotFound)` and `errors.As(err, &apiErr)` only work if every layer wrapped with `%w`.\n- Define and return sentinel errors (`var ErrNotFound = errors.New(\"order not found\")`) for conditions callers need to branch on; wrap them too (`fmt.Errorf(\"...: %w\", ErrNotFound)`).\n- Don't double-wrap into noise: one `%w` per layer boundary that adds real context is right; wrapping at every single call adds nothing.\n- Check every error \u2014 no `_ = err` or empty `if err != nil {}` without a comment saying why it's intentionally dropped.\n\n**The two together**\n\nIf `client.Get` returns both a timeout and a value, or you need to translate a downstream error into a domain error, do that translation here at the boundary \u2014 check `errors.Is(err, context.DeadlineExceeded)` if you want a domain-specific timeout error, but still wrap with `%w` so the original is reachable.\n\nQuick check before you call it done: `go build ./... && go vet ./... && go test -race ./...` all green, and `gofmt -l` clean.",
1125
+ "outputSha256": "71dda9f9ec164d84e0795c8abb241c6d315437006a9bb1a5e754e16f0eda37aa",
1126
+ "promptSha256": "9712a2b3973edb8bd2f0db6bd2f4cda7d89edf826fc0f006925bf1189b24e2c3",
1127
+ "deterministic": [
1128
+ true
1129
+ ],
1130
+ "judge": {
1131
+ "verdict": "pass",
1132
+ "reason": "Criterion 1: answer shows `func (s *Service) Fetch(ctx context.Context, id string)` and says forward same ctx to downstream `s.client.Get(ctx, id)`, not stored on struct. Criterion 2: concrete wrap `return nil, fmt.Errorf(\"fetching order %s: %w\", id, err)` returned to caller. Criterion 3: connects to `errors.Is(err, order.ErrNotFound)` and `errors.As(err, &apiErr)`. No fail criteria: explicitly warns against `_ = err` and %v."
1133
+ },
1134
+ "passed": true
1135
+ },
1136
+ {
1137
+ "output": "For a function in `internal/order` that calls a downstream client, here's the shape to aim for:\n\n## Context\n\nThread the caller's `context.Context` through as the first parameter \u2014 don't create a new one inside your function, and don't store it on a struct field:\n\n```go\n// internal/order/service.go\nfunc (s *Service) Submit(ctx context.Context, o Order) error {\n // pass ctx straight down to every call that touches the network,\n // disk, or another blocking operation\n resp, err := s.client.Post(ctx, \"/orders\", o)\n if err != nil {\n return fmt.Errorf(\"submitting order %s: %w\", o.ID, err)\n }\n // ...\n _ = resp\n return nil\n}\n```\n\nKey points:\n\n- `ctx` originates at the outermost boundary (`r.Context()` in an HTTP handler, `signal.NotifyContext` in a CLI main) and flows down unchanged. `context.Background()` belongs only at true process boundaries like `main`, never inside a call chain \u2014 using it here would silently break any caller's cancellation, deadline, or tracing values.\n- If this function needs a *narrower* deadline than its caller, derive it locally with `context.WithTimeout(ctx, ...)` and `defer cancel()` immediately. Never widen or replace the caller's context.\n- Don't put the context on `Service`; that breaks cancellation and makes the value ambiguous across concurrent calls.\n\n## Errors\n\nWrap at the boundary where you add meaningful context, and preserve the chain with `%w`:\n\n```go\nresp, err := s.client.Post(ctx, \"/orders\", o)\nif err != nil {\n return fmt.Errorf(\"submitting order %s: %w\", o.ID, err)\n}\n```\n\n- `%w` (not `%v`) so callers can still do `errors.Is` / `errors.As` \u2014 the caller you can't see yet is exactly who that serves. `%w` costs nothing and keeps the option open.\n- Wrap once per layer, with a message that names *what* you were doing, not what the callee already said. Don't wrap the same error twice in the same function.\n- If the downstream failure is a distinct outcome the caller should branch on, translate it: define a sentinel (`var ErrOrderRejected = errors.New(...)`) or a typed error and return that, wrapping the original underneath. Compare with `errors.Is`/`errors.As` at the consumer, never by string matching.\n- Context cancellation is not a bug to wrap away \u2014 if you must intercept `ctx.Err()`, return it (or wrap it) so callers can still detect `context.Canceled` / `context.DeadlineExceeded` via `errors.Is`. Never swallow it.\n- If a downstream call returns a partial result plus an error, don't return a zero value and drop the error silently; return both or explicitly discard the value with a comment explaining why.\n- Never do an empty `if err != nil {}` or `_ = err` without a comment.\n\n## Downstream client contract\n\nIf `s.client` is an interface, define it at the consumer \u2014 i.e. right here in `internal/order`, sized to just the methods this package calls (likely a single `Post(ctx, path, body)`), not a giant interface mirrored from the client's own package. That keeps the dependency honest and avoids import-cycle pressure. Accept the interface, return concrete types.\n\n## Verify before calling it done\n\n```bash\ngo build ./...\ngo vet ./...\ngo test -race ./...\ngofmt -l internal/order # should print nothing\n```\n\nAnd check: every new error return is checked/wrapped, no goroutine is started without a join point, and no `context.Background()` crept into the call chain. Also confirm you're not using a stdlib feature newer than the `go` directive in `go.mod`.",
1138
+ "outputSha256": "157876e5bdca4af46440bfde8fc6c7de68f427d94e47544a984ec771c0262bec",
1139
+ "promptSha256": "9712a2b3973edb8bd2f0db6bd2f4cda7d89edf826fc0f006925bf1189b24e2c3",
1140
+ "deterministic": [
1141
+ true
1142
+ ],
1143
+ "judge": {
1144
+ "verdict": "pass",
1145
+ "reason": "Criterion 1: 'Thread the caller's context.Context through as the first parameter... don't store it on a struct field' with func Submit(ctx context.Context, o Order) passing ctx to s.client.Post(ctx,...). Criterion 2: concrete fix shown: return fmt.Errorf(\"submitting order %s: %w\", o.ID, err). Criterion 3: '%w (not %v) so callers can still do errors.Is / errors.As'. No fail criteria met."
1146
+ },
1147
+ "passed": true
1148
+ },
1149
+ {
1150
+ "output": "For a new function in `internal/order` calling a downstream client:\n\n**Pass context as the first parameter, never store it.** Signature shape:\n\n```go\nfunc (s *Service) Reserve(ctx context.Context, orderID string) (*Reservation, error)\n```\n\nIf that call fans out to more than one downstream call, keep forwarding the same `ctx` \u2014 don't reach for `context.Background()` anywhere inside `internal/order`; the caller's cancellation, deadline, and values must survive the whole chain. If and only if *your* function needs a tighter bound than the caller gave you (e.g. this particular downstream must fail faster than the caller's overall deadline), derive one:\n\n```go\nctx, cancel := context.WithTimeout(ctx, s.reserveTimeout)\ndefer cancel()\n```\n\n`defer cancel()` is mandatory at every `With*` site \u2014 dropping it leaks the timer and the child context.\n\n**Do not add a `ctx` parameter for the sake of it if the client call itself is what needs it** \u2014 but in Go, threading it through your own public method is the norm even when today it's just forwarded, because the caller who wraps you in a deadline tomorrow shouldn't need to see a signature change.\n\n**Errors:** wrap at this layer's boundary with the operation's name, using `%w`:\n\n```go\nres, err := s.client.Reserve(ctx, req)\nif err != nil {\n return nil, fmt.Errorf(\"reserving order %s: %w\", orderID, err)\n}\n```\n\n`%w` (not `%v`) so the caller can `errors.Is` against a sentinel the downstream returns (e.g. a `client.ErrNotFound`) or `errors.As` a typed error. Don't double-log: wrap-and-return here, and let the layer that decides the operation failed (the HTTP handler, the CLI) do the `slog` call with structured fields \u2014 avoid both wrapping *and* logging the same error at the same layer.\n\nIf this function starts multiple goroutines (e.g. reserving several downstreams concurrently), use an `errgroup.Group` so the first error cancels the rest via your forwarded `ctx`, and join before returning:\n\n```go\ng, ctx := errgroup.WithContext(ctx)\nfor _, id := range ids {\n g.Go(func() error {\n return s.reserveOne(ctx, id)\n })\n}\nif err := g.Wait(); err != nil {\n return nil, fmt.Errorf(\"reserving batch: %w\", err)\n}\n```\n\nNo unjoined goroutines, no stored contexts, no swallowed errors. Verify with `go build ./...`, `go vet ./...`, and `go test -race ./...` before calling it done.",
1151
+ "outputSha256": "5a2de504d1dde5b5eb1ea1a98b23bf89215b74334be1decfd54e05eb8a8780f3",
1152
+ "promptSha256": "9712a2b3973edb8bd2f0db6bd2f4cda7d89edf826fc0f006925bf1189b24e2c3",
1153
+ "deterministic": [
1154
+ true
1155
+ ],
1156
+ "judge": {
1157
+ "verdict": "pass",
1158
+ "reason": "Criterion 1: signature `func (s *Service) Reserve(ctx context.Context, orderID string)` and 'never store it', forwarding same ctx. Criterion 2: concrete `return nil, fmt.Errorf(\"reserving order %s: %w\", orderID, err)` wrapping and returning. Criterion 3: '%w (not %v) so the caller can errors.Is ... or errors.As a typed error'. No fail criteria met."
1159
+ },
1160
+ "passed": true
1161
+ },
1162
+ {
1163
+ "output": "For a new function in `internal/order` that calls a downstream client, here is how to handle context and errors.\n\n## Context\n\n1. **Take `ctx context.Context` as the first parameter.** Never store it on a struct; thread it through explicitly. If `OrderService` is a struct, the method signature should be:\n\n ```go\n func (s *OrderService) Submit(ctx context.Context, req SubmitRequest) (*SubmitResult, error)\n ```\n\n2. **Forward the same `ctx` to the client call.** Do not substitute `context.Background()` or `context.TODO()` anywhere in the chain \u2014 that silently severs cancellation, deadlines, and values originating from the caller (an HTTP handler's `r.Context()`, a CLI's `signal.NotifyContext`, etc.).\n\n3. **Add a timeout at the boundary you own.** The downstream client call is an outbound dependency; if the caller's deadline is unbounded, wrap it here rather than relying on the client's own default:\n\n ```go\n ctx, cancel := context.WithTimeout(ctx, s.callTimeout)\n defer cancel()\n ```\n\n `defer cancel()` is mandatory on every `WithTimeout`/`WithCancel`/`WithDeadline` result \u2014 a missing `cancel` is a context leak. Because you derived a child, cancelling it does not affect the caller's parent context.\n\n4. **If you spawn a goroutine** to fan out (e.g. calling several downstream endpoints), give it a join point and a cancellation path: use `errgroup.Group` when you need the first error plus shared cancellation, or `sync.WaitGroup` (or `WaitGroup.Go` if `go.mod`'s `go` directive is >= 1.25) when you only need \"all finished\". Every goroutine needs a known owner and exit condition \u2014 no bare `go func() {}()`.\n\n## Errors\n\n1. **Wrap with `%w`, describing the operation, not the error.**\n\n ```go\n resp, err := s.client.Fetch(ctx, id)\n if err != nil {\n return nil, fmt.Errorf(\"fetching order %s: %w\", id, err)\n }\n ```\n\n Use `%w` (not `%v`) so callers can `errors.Is`/`errors.As`. The wrapping message supplies the \"what were we doing\"; the wrapped error supplies the \"why\".\n\n2. **Wrap once per meaningful layer boundary**, not at every call. Wrapping the same error five times in one function produces unreadable chains and duplicated context. One wrap at the point where you cross from `internal/order` into the client (and again where a higher layer crosses its own boundary) is the right granularity.\n\n3. **Do not swallow errors.** No `_ = err` without a comment explaining why it is safe, and no empty `if err != nil {}`. Check it, wrap it, or return it.\n\n4. **Sentinel/typed errors are the contract the client package owns.** If the client returns `client.ErrNotFound` or a typed error, compare with `errors.Is`/`errors.As`, never string equality. If callers of your new function need to distinguish outcomes, expose your own sentinel (`var ErrOrderRejected = errors.New(...)`) and wrap with it, so its callers use `errors.Is(err, order.ErrOrderRejected)`.\n\n5. **Decide the boundary between \"retryable\" and \"terminal\" explicitly** if the downstream client is flaky. Retry logic belongs here, keyed off `errors.Is(err, context.DeadlineExceeded)`/`context.Canceled` and the client's own classification \u2014 and it must honor `ctx` (stop retrying once `ctx.Err() != nil`), never spin past the deadline.\n\n6. **Partial-failure results.** If the function returns a result alongside an error, return a zero/nil result with a non-nil error rather than a half-populated value \u2014 callers should not have to inspect the result when `err != nil`.\n\n## Before reporting done\n\nRun `go build ./...`, `go vet ./...`, and `go test -race ./...`, plus `gofmt -l` for formatting. The function is complete when every error return is checked/wrapped/discarded-with-reason, `ctx` flows unbroken from caller to client, and any derived context is `defer cancel()`-ed.",
1164
+ "outputSha256": "48e2f374b916ea7bfabdd82a619e44b27beb8a4e3075120a045c3ca281d66eb1",
1165
+ "promptSha256": "9712a2b3973edb8bd2f0db6bd2f4cda7d89edf826fc0f006925bf1189b24e2c3",
1166
+ "deterministic": [
1167
+ true
1168
+ ],
1169
+ "judge": {
1170
+ "verdict": "pass",
1171
+ "reason": "Criterion 1: answer states ctx must be first parameter, never stored on struct, and forwarded to client call (\"Forward the same ctx to the client call\"). Criterion 2: concrete fix shown: `return nil, fmt.Errorf(\"fetching order %s: %w\", id, err)` with %w not %v. Criterion 3: connects wrap to callers using errors.Is/errors.As. No fail criteria: explicitly warns against `_ = err` and %v."
1172
+ },
1173
+ "passed": true
1174
+ }
1175
+ ]
1176
+ },
1177
+ {
1178
+ "id": "goroutine-lifetime",
1179
+ "kind": "behavior",
1180
+ "prompt": "I want to start a few goroutines to process items concurrently in Go. What should I watch out for?",
1181
+ "strictness": "high",
1182
+ "trials": 10,
1183
+ "passes": 10,
1184
+ "passRate": 1,
1185
+ "passAtK": 1,
1186
+ "grader": "judge",
1187
+ "status": "ran",
1188
+ "trialRecords": [
1189
+ {
1190
+ "output": "Here's what to watch out for when fanning out goroutines over a set of items.\n\n**1. Every goroutine needs an owner and an exit path**\n\nA goroutine with no bound isn't \"done when it's done\" \u2014 the caller has to know when it's finished. Decide up front:\n\n- **Just need \"all finished\"?** `sync.WaitGroup`, or `wg.Go(fn)` if your `go.mod`'s `go` directive is >= 1.25 (check it \u2014 don't reach for `WaitGroup.Go` on a 1.22 module).\n- **Need the first error + shared cancellation?** `errgroup.Group` (with `errgroup.WithContext` if you want a derived `ctx` that cancels on the first failure). This is usually the right tool for \"process N items concurrently.\"\n- **Need a bounded result stream?** A channel plus a `WaitGroup`/`errgroup` to close it after all senders finish.\n\nWhatever you pick, someone must `Wait`/`Wait()` \u2014 an unjoined goroutine can still race with process shutdown or leak.\n\n**2. Bound your concurrency**\n\n\"Start a goroutine per item\" is fine for ten items and a disaster for a million. Either:\n\n- a worker pool: a fixed number of goroutines reading from a channel, or\n- a semaphore (`golang.org/x/sync/semaphore`, or a buffered-channel token pattern), acquired before each spawn.\n\nUnbounded fan-out is the single most common way this pattern takes down a service (fd exhaustion, memory blowup, hammering a downstream dependency).\n\n**3. Propagate `context.Context`; never use `context.Background()` mid-chain**\n\nThe goroutine should receive the caller's `ctx` as its first parameter (not a struct field). That gives you:\n\n- cancellation when the caller goes away (client disconnects, CLI interrupt),\n- deadlines inherited from upstream.\n\nIf one item fails and you want the rest to stop, `errgroup.WithContext` cancels the derived context for all siblings. If you *don't* want that, use plain `errgroup.Group` \u2014 but then be explicit about it, because the default surprise is \"one failure cancelled everything.\"\n\n**4. Loop-variable capture** \u2014 a footnote now, but worth knowing\n\nSince Go 1.22, loop variables are per-iteration, so `for _, item := range items { go func() { use(item) }() }` is correct on a 1.22+ module. On an older module, pass the variable in as a parameter: `go func(item Item) { ... }(item)`. Check the `go` directive before assuming.\n\n**5. Error handling: collect or fail-fast, not \"log and forget\"**\n\n`errgroup` gives you the first error via `Wait()`. If you want *all* errors, have each worker write to a slice under a mutex, or to an error channel \u2014 don't `_ = err` or log-and-swallow, because the caller then thinks the batch succeeded. Wrap with `%w` so `errors.Is`/`errors.As` still work at the call site.\n\n**6. Don't write to a shared structure without synchronization**\n\nConcurrent `append` to the same slice, writes to a shared `map`, or incrementing a shared counter are data races \u2014 `go test -race ./...` will find them. Either guard with a `sync.Mutex`, have each worker own its own slice and merge after `Wait()`, or funnel results through a channel to a single collector goroutine. (A single goroutine is often cleaner than a mutex here.)\n\n**7. Close channels from the sender side, exactly once**\n\nIf workers send on a channel, the pattern is: workers send, a `WaitGroup` tracks them, and one goroutine does `wg.Wait(); close(ch)`. Never close from the receiver, and never close from multiple workers. The receiver ranges over the channel and naturally stops on close.\n\n**8. Side effects aren't ordered \u2014 make sure that's okay**\n\nConcurrent workers complete in nondeterministic order. If items share a resource (a file, a database row, an ID sequence), either serialize that part, use a per-item key, or accept/define the ordering. \"It seemed to work on my machine\" usually means the race just hasn't fired yet.\n\n**Sketch of the common shape:**\n\n```go\nfunc processAll(ctx context.Context, items []Item, workers int) error {\n g, ctx := errgroup.WithContext(ctx)\n ch := make(chan Item)\n\n for range workers {\n g.Go(func() error {\n for it := range ch {\n if err := processOne(ctx, it); err != nil {\n return fmt.Errorf(\"process %s: %w\", it.ID, err)\n }\n }\n return nil\n })\n }\n\n g.Go(func() error {\n defer close(ch)\n for _, it := range items {\n select {\n case ch <- it:\n case <-ctx.Done():\n return ctx.Err()\n }\n }\n return nil\n })\n\n return g.Wait()\n}\n```\n\nThen verify: `go build ./...`, `go vet ./...`, `go test -race ./...`. The race detector is your best friend for everything in points 6 and 8 \u2014 run it before you call the change done.",
1191
+ "outputSha256": "e9f01e1e91ea5a6a79502b02b44b1bfd73e0be9c8727667ec2342d12aa9364c9",
1192
+ "promptSha256": "af9e8082f58896d17ec3ae25f4b15493f270cc11a0beaafe987a65f270493dd4",
1193
+ "deterministic": [],
1194
+ "judge": {
1195
+ "verdict": "pass",
1196
+ "reason": "Names errgroup.Group/sync.WaitGroup/wg.Go/done channel (pt1). Says decide up front: 'Decide up front' owner/exit path, ctx cancellation via errgroup.WithContext (pt2). Propagates errors: 'errgroup gives you the first error via Wait()', 'don't log-and-swallow', %w wrapping (pt3). No fail criteria: explicitly warns against log-and-forget and unjoined goroutines."
1197
+ },
1198
+ "passed": true
1199
+ },
1200
+ {
1201
+ "output": "The main things to get right with concurrent goroutines in Go:\n\n**Ownership and exit** \u2014 every goroutine needs a known owner and a definite exit condition. An unjoined goroutine can outlive the caller, leak, or race with process shutdown. Decide up front how you'll wait for it.\n\n**Pick the right join primitive:**\n- `sync.WaitGroup` when you just need \"all finished\" and the individual errors don't matter or are handled inside each goroutine.\n- `errgroup.Group` (`golang.org/x/sync/errgroup`) when you want the first error surfaced plus shared cancellation. This is usually the better fit for a worker pool processing items, because one failure can cancel the rest.\n- If the project's `go.mod` targets Go 1.25 or later, `WaitGroup.Go` is available and avoids the manual `Add`/`Done` dance and its classic off-by-one bugs.\n\n**Context propagation** \u2014 if any goroutine does I/O or has a timeout expectation, thread the caller's `context.Context` in as the first parameter and forward it. Use `errgroup.WithContext` so one goroutine's error cancels siblings. Never stash a `ctx` on a struct field, and don't reach for `context.Background()` inside a call chain \u2014 that silently breaks cancellation and deadlines from the real caller.\n\n**Bounding concurrency for a worker pool** \u2014 don't spawn one goroutine per item if item count is unbounded. Cap it: a fixed pool of N workers reading from a channel, a buffered semaphore channel, or `errgroup` with `SetLimit(n)` (Go 1.20+). Unbounded goroutine spawning under load is a common way to run out of memory or file descriptors.\n\n**Shared state** \u2014 each goroutine writing to a shared map or slice needs a `sync.Mutex`, an atomic, or better: have workers send results back over a channel and let the collector be the single writer. Channels that stay single-writer are much easier to reason about. If you use a `sync.Mutex`, protect every access, not just writes.\n\n**Error handling inside workers** \u2014 check every error; don't swallow it with `_ = err` or an empty `if err != nil {}`. Wrap with `%w` so callers can `errors.Is`/`errors.As`, and be aware that with `errgroup` only the first error is returned unless you collect them yourself (e.g. `errors.Join`).\n\n**Loop variable capture** \u2014 this bit people for years. Since Go 1.22 each iteration has its own variable, so capturing the loop variable in a closure is safe; before 1.22 it wasn't. Match your `go.mod`'s declared `go` directive \u2014 if it targets 1.21 or older, you still need `i := i` inside the loop.\n\n**Panics in workers** \u2014 a panic in a goroutine that nobody recovers crashes the whole process. If a worker's work can panic on bad input, recover it and convert to an error, or make sure the code path genuinely can't panic.\n\n**Don't mutate a slice/map while iterating it** from another goroutine \u2014 snapshot or use locking.\n\n**Freeing resources** \u2014 any goroutine holding a channel, connection, or file should release it on all exit paths, including the error/cancel path. `defer` as usual.\n\n**Verification** \u2014 always run `go test -race ./...`; the race detector catches most of the above before production does. Combine with `go build ./...` and `go vet ./...`.\n\nA minimal errgroup pattern that covers most of these:\n\n```go\ng, ctx := errgroup.WithContext(ctx)\ng.SetLimit(8)\nfor _, item := range items {\n item := item // only needed if go.mod targets < 1.22\n g.Go(func() error {\n if err := process(ctx, item); err != nil {\n return fmt.Errorf(\"processing %s: %w\", item.ID, err)\n }\n return nil\n })\n}\nif err := g.Wait(); err != nil {\n return fmt.Errorf(\"processing items: %w\", err)\n}\n```\n\nThat gives you bounded concurrency, cancellation on first error, wrapped errors, and a single join point \u2014 which is the shape you want for almost any \"process these items concurrently\" task.",
1202
+ "outputSha256": "46a37ef8ae7a0ab8a2ee41927125315a2491f946cfe86271531d287f01f02c76",
1203
+ "promptSha256": "af9e8082f58896d17ec3ae25f4b15493f270cc11a0beaafe987a65f270493dd4",
1204
+ "deterministic": [],
1205
+ "judge": {
1206
+ "verdict": "pass",
1207
+ "reason": "Names errgroup.Group, sync.WaitGroup, WaitGroup.Go, done channel as join primitives (criterion 1). States ownership/exit condition and ctx cancellation decided up front (criterion 2). errgroup surfaces first error, wrapped with %w, g.Wait() returns it (criterion 3). No fail criteria: doesn't say goroutines need no join, and error propagation is recommended, not dropped."
1208
+ },
1209
+ "passed": true
1210
+ },
1211
+ {
1212
+ "output": "Here's the short version, then the details.\n\n**The four things that bite people:**\n\n1. **No join point.** A goroutine with no one waiting on it can leak or race with process shutdown. Every goroutine needs an owner and a known exit.\n2. **Cancellation.** If the caller cancels or times out, your workers must notice and stop. Thread `context.Context` in.\n3. **Error handling.** \"First error wins\" vs \"collect all errors\" vs \"just log them\" \u2014 decide before you write it.\n4. **Shared state.** Concurrent writes to a map, slice, or plain struct field is a data race; `go test -race` will find it if you run it.\n\n**Pick the right primitive.**\n\n- `errgroup.Group` (from `golang.org/x/sync/errgroup`) when you need the first error and shared cancellation. This is the common case for \"fan out N workers, one failure aborts the rest.\"\n- `sync.WaitGroup` when you only need \"all finished\" and errors are collected some other way. If the module's `go` directive is >= 1.25, `WaitGroup.Go` saves the `Add`/`Done` boilerplate.\n- A bounded worker pool (channel of items, K workers reading from it) when the number of items is large or unbounded \u2014 do not spawn one goroutine per item and hope.\n\n**Concrete sketch of the errgroup pattern:**\n\n```go\nfunc Process(ctx context.Context, items []Item) error {\n g, ctx := errgroup.WithContext(ctx)\n g.SetLimit(8) // bound concurrency\n\n for _, it := range items {\n it := it\n g.Go(func() error {\n if err := process(ctx, it); err != nil {\n return fmt.Errorf(\"processing %s: %w\", it.ID, err)\n }\n return nil\n })\n }\n return g.Wait()\n}\n```\n\n`SetLimit` is the bounded-pool knob; `g.Wait()` is the join point; `g.WithContext` gives you a ctx that cancels the first time any worker returns an error. Note `it := it` is only needed before Go 1.22 \u2014 if the module targets 1.22+, loop variables are per-iteration and you can drop it.\n\n**Watch for these traps:**\n\n- **Writing to a captured loop variable** (pre-1.22) or to any shared `results` slice without a mutex or a per-index slot. Use `results[i] = ...` with a pre-sized slice, or a channel, or a `sync.Mutex`.\n- **`errgroup` with an unhandled error path** \u2014 if you return `nil` on failure intent, `Wait` returns nil and you think it succeeded.\n- **Context stored on a struct** instead of threaded as the first parameter. Do not do this.\n- **`context.Background()` in the middle of a chain.** It silently breaks cancellation propagation from the caller.\n- **Unbuffered channels in a pool with a `return` in the worker.** Workers block on send forever; the pool never drains. Buffer, or select on `ctx.Done()`.\n- **Swallowing errors:** `if err != nil {}` or `_ = err` with no comment. Check it, wrap it with `%w`, or discard it deliberately with a reason.\n- **`%v` instead of `%w`** when wrapping \u2014 costs nothing to preserve unwrapping, and the caller you haven't written yet is exactly who needs `errors.Is`/`errors.As`.\n- **Assuming the goroutine finished before the caller returns.** \"Short-lived\" is not \"guaranteed joined.\"\n\n**Before you call it done:**\n\n```bash\ngo build ./...\ngo vet ./...\ngo test -race ./...\n```\n\n`-race` is the whole point here \u2014 run it, don't skip it. And check the module's `go` directive before reaching for `WaitGroup.Go`; if `go.mod` says 1.23, that method doesn't exist for you yet.\n\nIf you tell me roughly what you're doing (fan-out with one error, long-lived worker pool, pipeline stages), I can narrow this to the specific pattern that fits.",
1213
+ "outputSha256": "b00f7736b2433d76b0f9373af62c5d7f8bc0a842f6126c4e64bd7da25f161c91",
1214
+ "promptSha256": "af9e8082f58896d17ec3ae25f4b15493f270cc11a0beaafe987a65f270493dd4",
1215
+ "deterministic": [],
1216
+ "judge": {
1217
+ "verdict": "pass",
1218
+ "reason": "Names errgroup.Group, sync.WaitGroup/WaitGroup.Go, and bounded worker pool as join mechanisms (criteria 1). States owner/exit and ctx cancellation must be decided before writing (points 1-3, 'decide before you write it'). Error propagation: errgroup returns first error, g.Wait() join, %w wrapping, warns against returning nil on failure. No fail criteria: explicitly rejects 'short-lived' assumption."
1219
+ },
1220
+ "passed": true
1221
+ },
1222
+ {
1223
+ "output": "A few things matter most when you fan out work across goroutines. The short version: every goroutine needs an owner, a join point, and a way to stop.\n\n**Give every goroutine a join and a cancel path**\n- Use `errgroup.Group` (from `golang.org/x/sync/errgroup`) when you want the first error back and shared cancellation \u2014 `g, ctx := errgroup.WithContext(ctx)`, then `g.Go(func() error {...})`, then `err := g.Wait()`. The derived `ctx` is cancelled as soon as any `g.Go` returns a non-nil error, so the rest stop early.\n- Use `sync.WaitGroup` when you just need \"all finished\" and errors are handled some other way. If the module's `go` directive is >= 1.25 you can use `wg.Go(func() { ... })`; otherwise the classic `wg.Add(1)` / `defer wg.Done()` pattern.\n- Never launch a bare `go f()` with no `WaitGroup`/`errgroup` and no `ctx`. \"It's short-lived\" is not the same as \"guaranteed to finish before the caller returns\" \u2014 an unjoined goroutine can race with process shutdown or leak.\n\n**Propagate `context.Context`**\n- Pass it as the first parameter down the whole call path; do not store it on a struct field.\n- The root should come from a real source \u2014 an HTTP handler's `r.Context()`, or `signal.NotifyContext` in `main` \u2014 not `context.Background()` buried inside internal code, which silently breaks cancellation and deadlines.\n- Inside a worker, check `ctx.Done()` / select on it so a cancelled parent actually stops the work.\n\n**Bound the concurrency**\n- Unbounded `go` per item is a footgun for large inputs. Use a fixed worker pool (N workers reading from a channel) or a semaphore (`golang.org/x/sync/semaphore` or a buffered channel of size N) to cap in-flight work.\n\n**Share data carefully**\n- Prefer channels or a mutex over shared mutable state. `go test -race ./...` is your proof \u2014 run it, don't assume correctness.\n- If workers write to a shared slice/map, either have each worker return its result and collect it in the single goroutine that calls `Wait()`, or guard the writes with a `sync.Mutex`. Don't append to one slice from many goroutines without synchronization.\n\n**Handle errors, don't drop them**\n- Collect errors from workers (errgroup gives you the first; a slice guarded by a mutex gives you all) rather than `_ = err` or an empty `if err != nil {}`.\n- Wrap with `fmt.Errorf(\"processing item %d: %w\", i, err)` so callers can still use `errors.Is`/`errors.As`.\n\n**Watch the goroutine-per-channel pattern**\n- If you spawn a producer goroutine that writes to a channel, close the channel when the producer finishes (and only the sender closes it), and make sure consumers can exit on `ctx.Done()` \u2014 otherwise a blocked send on a channel nobody reads leaks the producer forever.\n\nMinimal sketch:\n\n```go\nfunc processAll(ctx context.Context, items []Item) error {\n g, ctx := errgroup.WithContext(ctx)\n sem := make(chan struct{}, 8) // cap concurrency\n for _, it := range items {\n it := it\n g.Go(func() error {\n select {\n case sem <- struct{}{}:\n defer func() { <-sem }()\n case <-ctx.Done():\n return ctx.Err()\n }\n return process(ctx, it)\n })\n }\n return g.Wait()\n}\n```\n\nThen verify with `go build ./...`, `go vet ./...`, and `go test -race ./...`.",
1224
+ "outputSha256": "c46bab30b1bb74cdca586a55c1f91d80ce2210c86a1a2c8c1a32e54683dfc88f",
1225
+ "promptSha256": "af9e8082f58896d17ec3ae25f4b15493f270cc11a0beaafe987a65f270493dd4",
1226
+ "deterministic": [],
1227
+ "judge": {
1228
+ "verdict": "pass",
1229
+ "reason": "Names errgroup.Group (g.Go/g.Wait, first error + shared ctx cancellation), sync.WaitGroup/WaitGroup.Go, and done-channel/ctx.Done exit. Says decide join/cancel before starting ('every goroutine needs an owner, a join point, and a way to stop'; 'Never launch a bare go f()'). Propagates errors via errgroup first error and wrapping with %w, not dropping. No fail criteria met."
1230
+ },
1231
+ "passed": true
1232
+ },
1233
+ {
1234
+ "output": "Process items concurrently in Go \u2014 key things to watch out for:\n\n## 1. Every goroutine needs an owner and an exit path\n\nNever launch a goroutine with no way to know when it finishes and no way to stop it. That means a join point (`sync.WaitGroup`) and/or a cancellation path (`context.Context`). \"It's short-lived\" is not an exit condition.\n\n```go\nvar wg sync.WaitGroup\nfor _, item := range items {\n wg.Add(1)\n go func(item Item) {\n defer wg.Done()\n process(item)\n }(item)\n}\nwg.Wait()\n```\n\nIf the project's `go.mod` targets Go 1.25+, prefer `wg.Go(func() { ... })` \u2014 same result, less boilerplate, no lost `Done`.\n\n## 2. Loop-variable capture\n\nGo 1.22+ gives each iteration its own loop variable, so `go func() { use(item) }()` is safe on modern toolchains. If `go.mod` declares < 1.22, you must pass the variable in as a parameter (as above) or shadow it. Match the declared `go` directive rather than assuming.\n\n## 3. Use `errgroup` when you need errors and shared cancellation\n\nIf you want the first error plus cancellation of siblings on failure, reach for `golang.org/x/sync/errgroup` instead of hand-rolling a `WaitGroup` + error channel:\n\n```go\ng, ctx := errgroup.WithContext(ctx)\nfor _, item := range items {\n item := item\n g.Go(func() error {\n return process(ctx, item) // ctx cancels siblings if any returns an error\n })\n}\nif err := g.Wait(); err != nil { ... }\n```\n\nUse a plain `WaitGroup` only when you genuinely just need \"all finished\" and errors are handled per-item.\n\n## 4. Bound the concurrency\n\nSpawning one goroutine per item is fine for tens of items and a disaster for tens of thousands \u2014 thousands of goroutines each holding a file descriptor, DB connection, or buffered payload will exhaust resources or overwhelm a downstream service. Cap it with a buffered-channel semaphore or `errgroup.SetLimit(n)`:\n\n```go\ng.SetLimit(8) // at most 8 in flight\n```\n\n## 5. Shared state needs a lock or channel\n\nConcurrent writes to a map or shared slice will race (or panic with `concurrent map writes`). Either guard with `sync.Mutex`, use `sync.Map` only if access is genuinely contended/hot, or funnel results through a single channel consumed by one goroutine:\n\n```go\nresults := make(chan Result, len(items))\n// workers send; a single reader collects or closes+range\n```\n\nAlways run `go test -race ./...` \u2014 it catches the races you cannot eyeball.\n\n## 6. Context propagation\n\nAccept `ctx` as the first parameter of every function on the path and forward it. Do not use `context.Background()` mid-chain \u2014 that silently breaks cancellation and deadlines for callers who set them. Do not store a `Context` in a struct field.\n\n## 7. Respect cancellation inside the worker\n\nA long-running `process` that ignores `ctx` will not stop when the parent is cancelled. Check `ctx.Err()` at natural boundaries, or use `select` against `ctx.Done()` when waiting on channels:\n\n```go\nselect {\ncase <-ctx.Done():\n return ctx.Err()\ncase r := <-work:\n ...\n}\n```\n\n## 8. Error handling\n\nDo not swallow errors (`_ = err` or an empty `if err != nil {}`). Wrap at each meaningful layer with `fmt.Errorf(\"processing %s: %w\", item.ID, err)` so callers can use `errors.Is`/`errors.As`. Use `%w`, not `%v`, even if you cannot see a caller that unwraps yet.\n\n## 9. Panics in workers\n\nA panic in a goroutine crashes the whole process \u2014 it is not recoverable by the launcher's caller. If a worker can panic on malformed input, recover inside the goroutine and convert it to an error (and log it), rather than letting it take down the process.\n\n## 10. Closing channels\n\nOnly the sender closes a channel, and only once. Never close from the receiver, and never close a channel multiple senders might still write to. Let `WaitGroup`/`errgroup` coordinate completion, then close.\n\n---\n\nBefore you report done: `go build ./...`, `go vet ./...`, and `go test -race ./...` should all pass, every new goroutine should have a visible join/cancel path, and every touched error return should be checked or wrapped.",
1235
+ "outputSha256": "53b00d21fd066d0e2934ab01288879ff6f4101155bd6e96a758f4bc0b77a703a",
1236
+ "promptSha256": "af9e8082f58896d17ec3ae25f4b15493f270cc11a0beaafe987a65f270493dd4",
1237
+ "deterministic": [],
1238
+ "judge": {
1239
+ "verdict": "pass",
1240
+ "reason": "Criterion 1: names sync.WaitGroup (with wg.Add/Done/Wait code) and errgroup.Group (g.Go/g.Wait) as join mechanisms. Criterion 2: section 1 states every goroutine needs an owner/exit path and ctx cancellation, and 'It's short-lived' is not an exit condition. Criterion 3: section 3 shows errgroup returning first error via g.Wait() and ctx cancelling siblings. No fail criteria: it never says goroutines need no join, and section 8 explicitly forbids swallowing errors."
1241
+ },
1242
+ "passed": true
1243
+ },
1244
+ {
1245
+ "output": "Short answer: every goroutine needs an owner and an exit condition. If you can't point at the line that waits for it or cancels it, you have a leak waiting to happen. The usual failure mode isn't a crash \u2014 it's a goroutine that blocks forever on a send/receive after everything else has moved on, and `go test -race` never sees it because nothing touched shared memory.\n\n## Practical checklist\n\n**1. Pick errgroup vs WaitGroup deliberately**\n- `errgroup.Group` (from `golang.org/x/sync`) when you need the first error back *and* shared cancellation \u2014 this is the right default for \"process N items, stop early on failure.\" Use `errgroup.WithContext` so the first error cancels the rest.\n- `sync.WaitGroup` when you just need \"all finished, no error aggregation.\" On Go 1.25+ prefer `wg.Go(fn)` over `wg.Add(1); go func(){ defer wg.Done(); ... }()` \u2014 it removes the Add/Done bookkeeping error class entirely. Check the module's `go` directive first; don't reach for `WaitGroup.Go` if `go.mod` targets 1.23.\n\n**2. Thread `context.Context` to every goroutine**\nAccept it as the first param of the worker function, select on `ctx.Done()` in any blocking wait, and never store it on a struct field. This is what lets a cancelled request actually stop the workers instead of letting them grind through a dead batch.\n\n**3. Know your goroutine's exit path for each of these**\n- Sends on a channel: is there guaranteed to be a receiver? A goroutine blocked on an unbuffered send with no reader is the classic leak.\n- Receives: is the channel guaranteed to close? Use `for v := range ch` only when the sender's closure is guaranteed; otherwise `select` with `ctx.Done()`.\n- `time.Sleep`/timers inside a worker: use a `select` on `ctx.Done()` with a timer, not a bare `time.Sleep`, if cancellation matters.\n\n**4. Bound concurrency**\n`go func()` per item in a loop over an unbounded slice will spawn unbounded goroutines. Either a fixed worker pool reading from a channel, or `errgroup.SetLimit(n)` (Go 1.20+) to cap in-flight work. Unbounded goroutine spawn is a real DoS vector when the input size is user-controlled.\n\n**5. Don't share mutable state without synchronization**\n- Prefer \"each goroutine owns its slice element\" (`results[i] = ...` where `i` is the goroutine's own index \u2014 safe, no lock needed).\n- Otherwise a `sync.Mutex` guarding the append/write, or `sync/atomic` for simple counters.\n- Don't append to a shared slice from multiple goroutines \u2014 `append` has a read-modify-write on the slice header and also races on the backing array.\n- Always run `go test -race ./...` on the package.\n\n**6. Error handling inside workers**\nWrap and return errors (`fmt.Errorf(\"processing item %d: %w\", i, err)`), don't swallow them, and decide up front whether one item's failure aborts the batch (errgroup) or is collected and reported per-item. `errgroup` gives you only the first error \u2014 if you need all of them, collect into a mutex-guarded slice or a channel yourself.\n\n**7. Closing channels**\nThe rule is: the *sender* closes, never the receiver, and never a goroutine that might still send. Closing a channel that another goroutine will write to panics with \"send on closed channel.\"\n\n**8. Check your `go.mod` go directive**\nMatch it for any language/stdlib feature (generics, `errgroup.SetLimit`, `WaitGroup.Go`, `slog`). Don't use a feature newer than the module targets.\n\n## The verification gate\n\nBefore you call it done:\n- `go build ./...`, `go vet ./...`, `go test -race ./...` all pass (and `golangci-lint run ./...` if the project has a config).\n- For every `go` statement you added, you can name its join point (`wg.Wait()` / `g.Wait()`) or its cancellation path (`ctx`).\n- `gofmt -l` reports nothing.\n\nIf you want this grounded in a concrete shape, the pattern I'd reach for is a bounded worker pool with `errgroup`:\n\n```go\ng, ctx := errgroup.WithContext(ctx)\ng.SetLimit(runtime.NumCPU())\nfor i, item := range items {\n i, item := i, item\n g.Go(func() error {\n if err := ctx.Err(); err != nil {\n return err // already cancelled, skip\n }\n res, err := process(ctx, item)\n if err != nil {\n return fmt.Errorf(\"processing item %d: %w\", i, err)\n }\n results[i] = res // each goroutine writes its own index \u2014 no lock\n return nil\n })\n}\nif err := g.Wait(); err != nil {\n return nil, err\n}\n```\n\nThe two things that make that safe: `SetLimit` bounds concurrency, and `results[i]` is written by exactly one goroutine per index, so no mutex is needed. Swap the `results[i] =` for an `append` to a shared slice and you've reintroduced the race.",
1246
+ "outputSha256": "4d3d251ed06076d67f92e8ae49eb30956ade8c10ce12ef8385fdf168cfa1694c",
1247
+ "promptSha256": "af9e8082f58896d17ec3ae25f4b15493f270cc11a0beaafe987a65f270493dd4",
1248
+ "deterministic": [],
1249
+ "judge": {
1250
+ "verdict": "pass",
1251
+ "reason": "Names errgroup.Group and sync.WaitGroup/WaitGroup.Go as join mechanisms (checklist #1, code sample g.Wait()). States exit condition/ownership decided before start ('every goroutine needs an owner and an exit condition', 'decide up front'), threads ctx. Propagates errors: errgroup returns first error, wraps with %w, g.Wait() returns it. No fail criteria met."
1252
+ },
1253
+ "passed": true
1254
+ },
1255
+ {
1256
+ "output": "Plain text answer, no tools \u2014 here's what to watch out for when fanning out goroutines in Go.\n\nOwn every goroutine you start\nEvery goroutine needs a known owner and a defined exit. Two questions to answer before `go f()`: who waits for this to finish, and what makes it stop if the work is cancelled? If you can't answer both, you've got a leak waiting to happen (or a race with process shutdown).\n\nUse a join point\n- `errgroup.Group` (from `golang.org/x/sync/errgroup`) when you want the first error back and shared cancellation. `g.Go(func() error {...})` then `err := g.Wait()`. If the project's `go.mod` allows, use `errgroup.WithContext` so the first failure cancels the others.\n- `sync.WaitGroup` when you just need \"all finished\" and don't care about errors. In Go 1.25 there's `WaitGroup.Go`, otherwise `wg.Add(1)` + `defer wg.Done()` inside each goroutine. Add before you `go`, not inside it \u2014 adding inside is a race with `Wait`.\n- Pick based on the project's `go` directive in `go.mod`; don't use `WaitGroup.Go` if the module targets < 1.25.\n\nPropagate context\nThread the caller's `context.Context` as the first parameter into every goroutine's work, not `context.Background()`. Use it for cancellation and deadlines; if the pool should stop when the caller disconnects, `WithContext` or manual `ctx.Done()` checks are what make that happen. Never stash a `Context` on a struct field \u2014 pass it.\n\nBound concurrency and backpressure\n\"A few goroutines\" is fine, but decide the bound explicitly. Unbounded `go` per item is a classic way to blow up on large input. Options:\n- A fixed worker pool: N workers reading from a channel.\n- `errgroup.SetLimit(n)` for a simple cap without hand-rolling workers.\n- A semaphore channel (`chan struct{}`) sized N.\nWhichever you pick, make sure producers respect it \u2014 a producer that keeps pushing into an unbuffered/limited queue just blocks (that's fine, that's backpressure) but a producer that ignores the bound and spawns its own goroutines defeats it.\n\nShare memory carefully\nAnything written concurrently must be synchronized. Prefer sending results on a channel over sharing a map/slice behind a mutex when you can. If you do share:\n- Use `sync.Mutex`/`RWMutex` for maps and slices.\n- Consider that `sync.Map` is only worth it for specific access patterns (mostly-read, stable keys); a plain map + mutex is often clearer.\n- Never assume a value written by one goroutine is visible to another without synchronization. Always run `go test -race ./...` \u2014 the race detector catches what review misses.\n\nWatch the channel lifecycle\n- Someone must close a channel, and only the sender should. Closing from the receiver, or from multiple senders, is a bug.\n- If writers don't all finish before readers stop, you leak blocked goroutines. `errgroup`/`WaitGroup` + close-after-Wait is the usual pattern.\n- Buffered channels change the failure mode: a buffer doesn't remove the need for a join point, it just delays the block.\n\nError handling inside goroutines\nA goroutine that returns an error to nowhere is a silent failure. Route errors out through `errgroup`, an error channel (first-error or collect-all, pick deliberately), or `slog` with context. Don't `panic` in a worker unless you have a recover policy \u2014 an unrecovered panic in any goroutine takes down the whole process. If you need isolation, `defer recover()` and convert to an error/result.\n\nLogging and observability\nWith `log/slog`, include per-item correlation (item id, worker id) as structured fields, or you'll have interleaved logs with no way to tell which item failed. Don't log inside tight loops at Info level unless you mean it.\n\nVerify before calling it done\nRun and require clean output:\n- `go build ./...`\n- `go vet ./...`\n- `go test -race ./...` (this is the big one for concurrency)\n- `gofmt -l` showing nothing\nIf a `.golangci.yml` exists, `golangci-lint run ./...` and fix at the root cause.\n\nAlso: run the tests multiple times (`-count=10`, and consider `GOMAXPROCS` variation or `-race` under load). Concurrency bugs are probabilistic; a single green run proves little.\n\nQuick sanity checklist\n- Each goroutine has an owner and a stop condition.\n- One join point (WaitGroup or errgroup) covers them all.\n- Context threaded through, not Background.\n- Concurrency bound chosen on purpose, not accidentally unbounded.\n- Shared state synchronized; results preferably sent on channels.\n- Channel close happens once, by the sender, after all sends.\n- Errors surfaced, not dropped.\n- Panic policy decided (recover or acceptable to crash).\n- `-race` clean.\n\nIf you tell me the shape \u2014 item source, results destination, and whether you need the first error or all errors \u2014 I can sketch the specific pool (errgroup vs. WaitGroup) that fits.",
1257
+ "outputSha256": "3e69159bf82bff8c55d459efb77ecc46a71b14f6e7a591e62805a75d5ef6bed3",
1258
+ "promptSha256": "af9e8082f58896d17ec3ae25f4b15493f270cc11a0beaafe987a65f270493dd4",
1259
+ "deterministic": [],
1260
+ "judge": {
1261
+ "verdict": "pass",
1262
+ "reason": "Names errgroup.Group, sync.WaitGroup/WaitGroup.Go, done channel as join points (\"Use a join point\" section). States owner/exit decided before `go f()` (\"Two questions to answer before `go f()`\"). Propagates errors via errgroup first error and error channel, not dropped. No fail criteria: explicitly warns against no join point and against dropping errors."
1263
+ },
1264
+ "passed": true
1265
+ },
1266
+ {
1267
+ "output": "Before touching the code, read `go.mod`'s `go` directive and look at how the package you're in already launches goroutines \u2014 that tells you whether `errgroup` is already a dependency and whether `WaitGroup.Go` (Go 1.25) is even available to you.\n\nThen, the things that actually bite:\n\n**1. Every goroutine needs an owner and an exit condition, decided up front.**\n- If you need the first error and the whole batch to cancel on failure: `errgroup.Group` with `errgroup.WithContext(ctx)` \u2014 the returned context cancels as soon as any task returns non-nil, and `g.Wait()` joins them all.\n- If you just need \"all finished, no error to collect\": a `sync.WaitGroup`. If the module's `go` directive is >= 1.25, use `wg.Go(func(){...})` instead of the manual `wg.Add(1)` / `defer wg.Done()` dance \u2014 it's the exact same thing with the two most common bugs (forgetting `Add`, mismatching `Done`) designed out.\n- Either way: no goroutine without a visible join point. \"It's short-lived\" is not a guarantee it finishes before the caller returns or before the process shuts down.\n\n**2. Cancel, don't just wait, when work is unbounded.**\nIf the batch can be large or one item can hang, a `WaitGroup` alone waits forever. A `ctx` you actually cancel (`defer cancel()`) \u2014 whether from `errgroup.WithContext`, `context.WithCancel`, or `signal.NotifyContext` at the CLI edge \u2014 is what lets you bail out. The context must be threaded as the first parameter through every function on the call path; never stash it on a struct field.\n\n**3. Do not capture the loop variable.**\nSince Go 1.22 each loop iteration has its own variable, so `go func(){ ...use item... }()` is finally safe \u2014 but only if the module's `go` directive says >= 1.22. If it's older, you must still shadow: `item := item` before the closure. Check the `go` directive; don't assume.\n\n**4. Bound the concurrency.**\n`for _, item := range items { g.Go(process(item)) }` on ten million items spawns ten million goroutines. Use a fixed-size worker pool (`n` workers pulling from a channel) or a semaphore (`golang.org/x/sync/semaphore`, or a buffered channel of size `n`) when the input size isn't something you control or have proven small.\n\n**5. Errors from workers: don't let them vanish.**\nA goroutine that hits an error and just returns it into the void, or worse, logs it and continues as if the item succeeded, is the classic silent-failure bug. With `errgroup`, return it and it surfaces from `Wait`. If you're collecting per-item results, make the error part of the result struct. And any error you do wrap, wrap with `%w` (`fmt.Errorf(\"processing item %d: %w\", id, err)`) so the caller can still `errors.Is`/`errors.As` it \u2014 `%v` throws that away for the caller you can't see yet. Never `_ = err` without a comment saying why.\n\n**6. Protect shared state.**\nIf workers write to the same map, slice, or counter, guard it with a `sync.Mutex` or use `sync/atomic` / `golang.org/x/sync/singleflight`. `go test -race ./...` will catch most of this \u2014 run it, don't skip it. Plain `go test` passes right over data races.\n\n**7. Decide what \"process items\" does to order and backpressure.**\nIf results must come back in input order, fan-in through a `[]result` indexed by item index and assemble after `Wait`; a channel of results gives you completion order, not input order, and mixing the two up is a subtle correctness bug that tests often miss. If the workers feed a downstream stage, that stage's buffer size is your backpressure control \u2014 an unbuffered channel makes every worker block on the consumer.\n\n**8. Context is per-call, not per-struct.** Resist the temptation to make the goroutine-pool struct hold a `ctx` field so you don't have to pass it around. That breaks cancellation the moment two callers with different deadlines share the pool, and it's one of the few Go rules with essentially no exceptions.\n\nThe shape most of this collapses into, with `errgroup`:\n\n```go\nfunc ProcessAll(ctx context.Context, items []Item) error {\n g, ctx := errgroup.WithContext(ctx)\n g.SetLimit(8) // bound the pool; skip only if the input is proven small\n\n for _, item := range items {\n g.Go(func() error {\n if err := process(ctx, item); err != nil {\n return fmt.Errorf(\"processing item %d: %w\", item.ID, err)\n }\n return nil\n })\n }\n return g.Wait()\n}\n```\n\nVerify with `go build ./...`, `go vet ./...`, `go test -race ./...` all exit 0, plus `gofmt -l` clean, before you call it done.",
1268
+ "outputSha256": "1dbdd679bda59feb1549c16201d8c5a58ac8b4dd2dbe965960c3f5cbdeed5a53",
1269
+ "promptSha256": "af9e8082f58896d17ec3ae25f4b15493f270cc11a0beaafe987a65f270493dd4",
1270
+ "deterministic": [],
1271
+ "judge": {
1272
+ "verdict": "pass",
1273
+ "reason": "Names errgroup.Group/WithContext, sync.WaitGroup/WaitGroup.Go, and done-channel-style ctx cancellation as join mechanisms (points 1-2). States owner/exit condition decided up front (point 1). Point 5 propagates errors via errgroup Wait and %w wrapping. No fail criteria: explicitly rejects 'short-lived' assumption and logging-and-dropping errors."
1274
+ },
1275
+ "passed": true
1276
+ },
1277
+ {
1278
+ "output": "Concurrent goroutines are one of Go's best features, and also one of its easiest ways to leak resources or ship a race. Here's what to watch out for, in roughly the order things tend to bite.\n\n**1. Every goroutine needs an owner and an exit condition.**\nA goroutine you \"just start\" with `go process(item)` and never join is a goroutine whose lifetime you don't control. Ask up front: who waits for this? If the answer is nobody, you have a leak waiting to happen. Two standard patterns:\n\n- `errgroup.Group` (from `golang.org/x/sync/errgroup`) when you want the first error back and shared cancellation. `g.Go(func() error { ... })`, then `g.Wait()`.\n- `sync.WaitGroup` when you just need \"all of these finished\" and don't care about propagating the first error. If your module targets Go >= 1.25, `wg.Go(fn)` is the tidy form; otherwise `wg.Add(1)` before the `go` statement, `defer wg.Done()` inside.\n\nThe classic bug: `wg.Add(1)` *inside* the goroutine instead of before launching it. That's a race \u2014 `Wait()` can return before `Add` runs.\n\n**2. Bound the concurrency.**\n\"Process items concurrently\" with one goroutine per item is fine for ten items and disastrous for ten million. Use a semaphore (`make(chan struct{}, n)`), a worker pool reading from a channel, or `errgroup`'s `SetLimit(n)`. An unbounded fan-out is how you exhaust file descriptors, DB connections, or memory.\n\n**3. Propagate `context.Context` \u2014 don't substitute `context.Background()` inside the chain.**\nEach goroutine should get the caller's `ctx` (or a child via `context.WithCancel`/`WithTimeout`), and every blocking call inside it should respect that context. If you call `context.Background()` mid-chain you've silently severed cancellation, so a `ctx` cancel from above won't stop your workers. For \"cancel all workers as soon as one fails\", derive a child context and cancel it from the error path \u2014 `errgroup.WithContext` does exactly this for you.\n\n**4. Don't share mutable state without a plan.**\nThis is what `go test -race ./...` is for. Options, in rough order of preference: keep each goroutine's work local and communicate results over a channel; protect a shared structure with a `sync.Mutex`/`sync.RWMutex`; or use atomics for simple counters. What you must not do is touch a map or slice from multiple goroutines without synchronization \u2014 Go's built-in maps will actually crash the runtime (not just corrupt data) on concurrent writes.\n\n**5. Watch for goroutine leaks from unread channels.**\nIf workers send results on a channel and the consumer stops reading (e.g. it returned early on error), the senders block forever. Make sends `select` on `ctx.Done()` as well as the channel, or ensure the consumer always drains. Similarly, always `close()` channels you own, once, from the sender side \u2014 closing from the receiver is a bug.\n\n**6. Get error handling right across goroutines.**\nYou can't `return err` from a goroutine. Collect errors into a slice guarded by a mutex, or use `errgroup`, which handles this for you and returns the first non-nil error from `Wait()`. Don't swallow errors inside workers with `_ =`.\n\n**7. Don't capture a loop variable by reference if you're on an older Go.**\nIf your module's `go` directive is < 1.22, `for _, item := range items { go func() { use(item) }() }` captures the same `item` \u2014 pass it as a parameter: `go func(it Item) { use(it) }(item)`. On 1.22+ the per-iteration variable semantics fix this, but passing explicitly is still clearer and safe everywhere.\n\nA quick checklist before you ship:\n\n- Each goroutine has a join point (`WaitGroup`/`errgroup`) or an explicit cancellation path.\n- Concurrency is bounded.\n- `ctx` flows through, no `context.Background()` mid-chain.\n- Shared state is either absent, mutex/atomic-guarded, or moved over a channel.\n- Channels are closed once, by the sender, and sends respect `ctx`.\n- `go build ./...`, `go vet ./...`, and (critically) `go test -race ./...` all pass.\n\nIf you paste the actual snippet you're starting from, I can point at the specific spot that needs a `WaitGroup`, a context, or a lock.",
1279
+ "outputSha256": "0623995d70d2d9b92ecee80c5659bad4d5802589d4a39608fef20120336a837c",
1280
+ "promptSha256": "af9e8082f58896d17ec3ae25f4b15493f270cc11a0beaafe987a65f270493dd4",
1281
+ "deterministic": [],
1282
+ "judge": {
1283
+ "verdict": "pass",
1284
+ "reason": "Names errgroup.Group (g.Go/g.Wait) and sync.WaitGroup/WaitGroup.Go as join mechanisms (sections 1, 6). States ownership/exit condition decided up front and ctx cancellation via errgroup.WithContext/child ctx (sections 1, 3). Propagates errors: errgroup returns first non-nil error from Wait(), don't swallow with _= (section 6). No fail criteria: never says skip joining; explicitly warns against dropping errors."
1285
+ },
1286
+ "passed": true
1287
+ },
1288
+ {
1289
+ "output": "# Concurrent goroutine processing in Go \u2014 what to watch for\n\nHere's what actually bites people, in rough order of how often it causes real bugs:\n\n## 1. Every goroutine needs an owner and an exit path\n\nThe classic leak is `go func() { ... }()` with no join point and no way out. Before you write the `go` keyword, answer two questions:\n\n- **Who waits for this to finish?** (`sync.WaitGroup`, `errgroup.Group`, or a channel the caller consumes)\n- **What makes it stop early?** (a `context.Context` that gets cancelled, or a closed input channel)\n\nIf you can't answer both, you're not ready to spawn it. The rules for this skill are explicit: never start a goroutine with no join point or cancellation path.\n\n## 2. `errgroup` vs `WaitGroup` \u2014 pick deliberately\n\n```go\n// Need the first error AND shared cancellation on failure:\ng, ctx := errgroup.WithContext(ctx)\nfor _, item := range items {\n item := item // see loop-variable caveat below\n g.Go(func() error {\n return process(ctx, item)\n })\n}\nif err := g.Wait(); err != nil {\n return fmt.Errorf(\"processing batch: %w\", err)\n}\n```\n\n```go\n// Just need \"all done\", error handling done another way:\nvar wg sync.WaitGroup\nfor _, item := range items {\n wg.Add(1)\n go func(item Item) {\n defer wg.Done()\n // ...\n }(item)\n}\nwg.Wait()\n```\n\nNotes:\n- `errgroup.WithContext` cancels the derived context the moment the **first** goroutine returns non-nil. If you want *all* items attempted regardless of failures, don't use the derived ctx for the per-item calls (or collect errors into a slice behind a mutex instead).\n- On Go 1.25+ with a matching `go` directive, `WaitGroup.Go(fn)` replaces the `Add(1)` + `defer Done()` pair. **Do not use it if the module's `go.mod` targets an older version** \u2014 this skill's rule is to match the declared directive.\n\n## 3. Loop variables\n\nGo 1.22+ gives each iteration a fresh loop variable, so `go func() { use(item) }()` is now safe *if* the module targets 1.22+. If the `go` directive is older, you must copy: `item := item` before the `go`, or pass it as a parameter. Check `go.mod` before assuming either way \u2014 this is exactly the kind of thing that silently changes meaning between module versions.\n\n## 4. Bound your concurrency\n\n`for range items { go process(item) }` over 100k items spawns 100k goroutines. That's cheap-ish but not free, and it hammers whatever downstream resource you touch. Use a semaphore or a fixed worker pool:\n\n```go\nsem := make(chan struct{}, maxConcurrent)\nfor _, item := range items {\n sem <- struct{}{}\n g.Go(func() error {\n defer func() { <-sem }()\n return process(ctx, item)\n })\n}\n```\n\nChoose `maxConcurrent` based on the bottleneck \u2014 DB connections, HTTP client pool size, etc. \u2014 not arbitrarily.\n\n## 5. Context propagation\n\n- Pass `ctx` as the **first parameter** to every function on the path. Never store it on a struct field.\n- Don't use `context.Background()` in the middle of a call chain \u2014 that silently severs cancellation and deadlines from the caller. It belongs at process edges (`main`, `TestMain`, HTTP handler `r.Context()` is already fine).\n- Make sure `process` actually *checks* `ctx` \u2014 either via blocking calls that accept it, or a `select { case <-ctx.Done(): return ctx.Err(); ... }`.\n\n## 6. Data races on shared state\n\n`go test -race ./...` is non-negotiable here \u2014 run it, don't skip it. Beyond the obvious:\n\n- **Slice appends** from multiple goroutines race even when indices don't overlap, because of the underlying array's length/cap.\n- **Maps** are not safe for concurrent write at all; a guarded map needs a `sync.Mutex`/`RWMutex` around *every* access, reads included.\n- **`sync.Map`** is only worth it for append-only or disjoint-key workloads; otherwise a mutex + map is faster and clearer.\n- The idiomatic alternative is **channels-own-data**: one goroutine owns the state, others send it messages. Usually less error-prone than scattered locks.\n\n## 7. Channel gotchas\n\n- Only the **sender** closes; closing from the receiver or from multiple places panics.\n- A send on a full unbuffered channel blocks forever if nobody's receiving \u2014 pair with `select { case ch <- v: case <-ctx.Done(): }`.\n- Don't leak a goroutine blocked on a send that will never be received because the consumer exited early. The `ctx` select handles this.\n- `range ch` ends when the channel is closed \u2014 the producer must close after its last send, and the consumer should still respect `ctx`.\n\n## 8. Panics in a goroutine take down the process\n\nA panic in a spawned goroutine cannot be recovered by the caller's `defer recover()`. If a worker touches anything that might panic (a plugin, a parser, user-supplied code), put a `defer func() { if r := recover(); r != nil { ... } }()` inside the goroutine so one poisoned item doesn't kill the whole program.\n\n## 9. Shutdown ordering\n\nAt process shutdown, roughly: cancel the root context \u2192 wait for the `WaitGroup`/`errgroup` to return \u2192 *then* close resources the workers were using (DB pool, files, loggers). Closing resources first gives you `use of closed connection` errors in the logs that look like data bugs but aren't.\n\n## Checklist before you ship\n\n```bash\ngo build ./...\ngo vet ./...\ngo test -race ./...\ngofmt -l . # should print nothing\n```\n\nAnd by hand:\n\n- [ ] Every `go` statement has a visible join point and a cancellation path.\n- [ ] Every error return is checked, wrapped with `%w` (`fmt.Errorf(\"processing item %s: %w\", id, err)`), or discarded with a comment saying why.\n- [ ] No `context.Context` stored on a struct; it's the first parameter everywhere.\n- [ ] Concurrency is bounded to something the downstream can absorb.\n- [ ] `go test -race` passes on a real workload, not just a trivial one.\n\nIf all of that holds, you're in good shape. The failure mode to keep in your head is the *silent* one: a goroutine that leaks, a context that stops propagating, a race that only manifests under load. `-race` and a bounded, joinable design are what catch those.",
1290
+ "outputSha256": "2c466c05bfea146d3de78360f1c3163a785ab542a7c005cb02b6d8c62a68b497",
1291
+ "promptSha256": "af9e8082f58896d17ec3ae25f4b15493f270cc11a0beaafe987a65f270493dd4",
1292
+ "deterministic": [],
1293
+ "judge": {
1294
+ "verdict": "pass",
1295
+ "reason": "Criterion 1: names errgroup.Group, sync.WaitGroup, WaitGroup.Go, and channels as join mechanisms with code. Criterion 2: \u00a71 says decide owner/exit path and ctx cancellation before writing 'go'. Criterion 3: errgroup example returns first error via g.Wait() and wraps with %w. No fail criteria: no short-lived justification, no log-and-drop recommendation."
1296
+ },
1297
+ "passed": true
1298
+ }
1299
+ ]
1300
+ }
1301
+ ],
1302
+ "verdict": "pass",
1303
+ "scope": "bundled",
1304
+ "skillDigest": "98b4c5ebfca41758ff617d21a5bd124de179e05ffa6d66363e51743259c240ab",
1305
+ "catalogDigest": "4f4016d410837e394a27e5b247e38ef2f57a1ee0baba4436ce7d3d71e223333d",
1306
+ "judgePromptVersion": "2026-09-25.1",
1307
+ "runner": "deepseek",
1308
+ "model": "deepseek-chat",
1309
+ "runnerPromptVersion": "2026-09-25.1",
1310
+ "recordedAt": "2026-09-25T05:09:44.123Z",
1311
+ "judge": "deepseek",
1312
+ "judgeModel": "deepseek-chat"
1313
+ },
1314
+ {
1315
+ "schemaVersion": "1.0.0",
1316
+ "skillId": "go/go-testing",
1317
+ "strictness": "high",
1318
+ "trials": 10,
1319
+ "triggerAccuracy": {
1320
+ "truePositive": 6,
1321
+ "falsePositive": 0,
1322
+ "positives": 6,
1323
+ "negatives": 6
1324
+ },
1325
+ "evidence": "authored",
1326
+ "scenarios": [
1327
+ {
1328
+ "id": "trigger-positive-1",
1329
+ "kind": "trigger-positive",
1330
+ "prompt": "Write table-driven Go tests for this parser function",
1331
+ "strictness": "high",
1332
+ "trials": 1,
1333
+ "passes": 1,
1334
+ "passRate": 1,
1335
+ "passAtK": 1,
1336
+ "grader": "trigger-rank-fork-family",
1337
+ "status": "ran",
1338
+ "deterministic": true
1339
+ },
1340
+ {
1341
+ "id": "trigger-positive-2",
1342
+ "kind": "trigger-positive",
1343
+ "prompt": "Add go test coverage for this package using t.Run subtests",
1344
+ "strictness": "high",
1345
+ "trials": 1,
1346
+ "passes": 1,
1347
+ "passRate": 1,
1348
+ "passAtK": 1,
1349
+ "grader": "trigger-rank-fork-family",
1350
+ "status": "ran",
1351
+ "deterministic": true
1352
+ },
1353
+ {
1354
+ "id": "trigger-positive-3",
1355
+ "kind": "trigger-positive",
1356
+ "prompt": "Fix this failing go test in the worker package",
1357
+ "strictness": "high",
1358
+ "trials": 1,
1359
+ "passes": 1,
1360
+ "passRate": 1,
1361
+ "passAtK": 1,
1362
+ "grader": "trigger-rank-fork-family",
1363
+ "status": "ran",
1364
+ "deterministic": true
1365
+ },
1366
+ {
1367
+ "id": "trigger-positive-4",
1368
+ "kind": "trigger-positive",
1369
+ "prompt": "Add a fuzz test for this Go decoder function",
1370
+ "strictness": "high",
1371
+ "trials": 1,
1372
+ "passes": 1,
1373
+ "passRate": 1,
1374
+ "passAtK": 1,
1375
+ "grader": "trigger-rank-fork-family",
1376
+ "status": "ran",
1377
+ "deterministic": true
1378
+ },
1379
+ {
1380
+ "id": "trigger-positive-5",
1381
+ "kind": "trigger-positive",
1382
+ "prompt": "Write a benchmark for this function using b.Loop",
1383
+ "strictness": "high",
1384
+ "trials": 1,
1385
+ "passes": 1,
1386
+ "passRate": 1,
1387
+ "passAtK": 1,
1388
+ "grader": "trigger-rank-fork-family",
1389
+ "status": "ran",
1390
+ "deterministic": true
1391
+ },
1392
+ {
1393
+ "id": "trigger-positive-6",
1394
+ "kind": "trigger-positive",
1395
+ "prompt": "Test this Go HTTP handler with httptest",
1396
+ "strictness": "high",
1397
+ "trials": 1,
1398
+ "passes": 1,
1399
+ "passRate": 1,
1400
+ "passAtK": 1,
1401
+ "grader": "trigger-rank-fork-family",
1402
+ "status": "ran",
1403
+ "deterministic": true
1404
+ },
1405
+ {
1406
+ "id": "trigger-negative-1",
1407
+ "kind": "trigger-negative",
1408
+ "prompt": "Write pytest tests for this Python function",
1409
+ "strictness": "high",
1410
+ "trials": 1,
1411
+ "passes": 1,
1412
+ "passRate": 1,
1413
+ "passAtK": 1,
1414
+ "grader": "trigger-rank-fork-family",
1415
+ "status": "ran",
1416
+ "deterministic": true
1417
+ },
1418
+ {
1419
+ "id": "trigger-negative-2",
1420
+ "kind": "trigger-negative",
1421
+ "prompt": "Add Jest tests for this React component",
1422
+ "strictness": "high",
1423
+ "trials": 1,
1424
+ "passes": 1,
1425
+ "passRate": 1,
1426
+ "passAtK": 1,
1427
+ "grader": "trigger-rank-fork-family",
1428
+ "status": "ran",
1429
+ "deterministic": true
1430
+ },
1431
+ {
1432
+ "id": "trigger-negative-3",
1433
+ "kind": "trigger-negative",
1434
+ "prompt": "Review this Go diff for goroutine leaks and races",
1435
+ "strictness": "high",
1436
+ "trials": 1,
1437
+ "passes": 1,
1438
+ "passRate": 1,
1439
+ "passAtK": 1,
1440
+ "grader": "trigger-rank-fork-family",
1441
+ "status": "ran",
1442
+ "deterministic": true
1443
+ },
1444
+ {
1445
+ "id": "trigger-negative-4",
1446
+ "kind": "trigger-negative",
1447
+ "prompt": "Fix the go build error in this module",
1448
+ "strictness": "high",
1449
+ "trials": 1,
1450
+ "passes": 1,
1451
+ "passRate": 1,
1452
+ "passAtK": 1,
1453
+ "grader": "trigger-rank-fork-family",
1454
+ "status": "ran",
1455
+ "deterministic": true
1456
+ },
1457
+ {
1458
+ "id": "trigger-negative-5",
1459
+ "kind": "trigger-negative",
1460
+ "prompt": "Implement a new Go service that calls this downstream API",
1461
+ "strictness": "high",
1462
+ "trials": 1,
1463
+ "passes": 1,
1464
+ "passRate": 1,
1465
+ "passAtK": 1,
1466
+ "grader": "trigger-rank-fork-family",
1467
+ "status": "ran",
1468
+ "deterministic": true
1469
+ },
1470
+ {
1471
+ "id": "trigger-negative-6",
1472
+ "kind": "trigger-negative",
1473
+ "prompt": "Write Rust unit tests for this crate",
1474
+ "strictness": "high",
1475
+ "trials": 1,
1476
+ "passes": 1,
1477
+ "passRate": 1,
1478
+ "passAtK": 1,
1479
+ "grader": "trigger-rank-fork-family",
1480
+ "status": "ran",
1481
+ "deterministic": true
1482
+ },
1483
+ {
1484
+ "id": "table-driven-subtests",
1485
+ "kind": "behavior",
1486
+ "prompt": "I need to add several test cases for this Go function that validates an order. How should I structure them?",
1487
+ "strictness": "high",
1488
+ "trials": 10,
1489
+ "passes": 8,
1490
+ "passRate": 0.8,
1491
+ "passAtK": 1,
1492
+ "grader": "judge",
1493
+ "status": "ran",
1494
+ "trialRecords": [
1495
+ {
1496
+ "output": "To structure those test cases well, I need to know a couple of things first:\n\n1. **What does the validation function look like?** Signature, return type, and what it considers \"valid\" vs. \"invalid.\" I can only see your description, not the code.\n2. **What's in your `go.mod`?** The `go` directive tells me whether the project's toolchain supports `t.Context()` and `b.Loop()` (both Go 1.24+). That changes how I write setup/teardown and benchmarks.\n3. **What conventions does the package already use?** Look at 1\u20132 existing `_test.go` files in the same package: the table struct shape, field naming (`want`/`wantErr` vs. something else), whether an assertion library is already in play, and whether tests use `t.Parallel()`.\n\nAssuming a typical validator like `func ValidateOrder(o Order) error`, here's the structure I'd use:\n\n```go\nfunc TestValidateOrder(t *testing.T) {\n\tt.Parallel()\n\n\ttests := []struct {\n\t\tname string\n\t\torder Order\n\t\twantErr error // or a bool + errors.Is check\n\t}{\n\t\t{\n\t\t\tname: \"valid order\",\n\t\t\torder: Order{ID: \"o-1\", Items: []Item{{SKU: \"a\", Qty: 1}}},\n\t\t\twantErr: nil,\n\t\t},\n\t\t{\n\t\t\tname: \"empty items\",\n\t\t\torder: Order{ID: \"o-2\"},\n\t\t\twantErr: ErrNoItems,\n\t\t},\n\t\t{\n\t\t\tname: \"zero-value order\",\n\t\t\torder: Order{},\n\t\t\twantErr: ErrMissingID,\n\t\t},\n\t\t{\n\t\t\tname: \"negative quantity\",\n\t\t\torder: Order{ID: \"o-3\", Items: []Item{{SKU: \"a\", Qty: -1}}},\n\t\t\twantErr: ErrBadQty,\n\t\t},\n\t\t// boundary: max allowed items, exactly-at-limit qty, etc.\n\t}\n\n\tfor _, tt := range tests {\n\t\tt.Run(tt.name, func(t *testing.T) {\n\t\t\tt.Parallel()\n\n\t\t\terr := ValidateOrder(tt.order)\n\t\t\tif tt.wantErr == nil {\n\t\t\t\tif err != nil {\n\t\t\t\t\tt.Fatalf(\"ValidateOrder() = %v, want nil\", err)\n\t\t\t\t}\n\t\t\t\treturn\n\t\t\t}\n\t\t\tif !errors.Is(err, tt.wantErr) {\n\t\t\t\tt.Fatalf(\"ValidateOrder() error = %v, want %v\", err, tt.wantErr)\n\t\t\t}\n\t\t})\n\t}\n}\n```\n\nKey points on structuring the cases:\n\n- **One struct per case** with `name` first, then inputs, then `want`/`wantErr`. Keep `t.Run(tt.name, ...)` always \u2014 it makes failures point at the exact case.\n- **Cover the categories, not just the happy path:** valid input, nil/empty/zero-value inputs, each distinct error condition, and boundary values (at-limit, one-over-limit).\n- **Assert errors with `errors.Is`/`errors.As`, never string matching.** That's why `wantErr` is typed as the sentinel `error` above rather than a `string`.\n- **Use `t.Helper()`** as the first line of any shared setup helper, and `t.Cleanup(...)` for teardown rather than scattered `defer`s.\n- **If validation reads a context,** use `t.Context()` (Go 1.24+) instead of hand-rolling `context.WithCancel` + `defer cancel()`.\n- **Put any fixture orders in `testdata/`**, named for the case that uses them, if they're large.\n\nIf you paste the function signature and one existing `_test.go` from the package, I can match your exact table/fixture conventions and name the real sentinel errors instead of the placeholders above.",
1497
+ "outputSha256": "9535793042f37cde6bfbd9052823e3c2f985c63efe9331bfd1f6b56b1d6b9a38",
1498
+ "promptSha256": "06210d5c296a9a06974e208465f448084532592fac5ad8558545416a0dc9572d",
1499
+ "deterministic": [],
1500
+ "judge": {
1501
+ "verdict": "pass",
1502
+ "reason": "Answer shows a slice-of-struct table with name/order/wantErr fields (tests := []struct{...}), loops with t.Run(tt.name, ...), and checks the specific error via errors.Is(err, tt.wantErr) with wantErr typed as sentinel error, not a boolean flag. All three pass criteria concretely present."
1503
+ },
1504
+ "passed": true
1505
+ },
1506
+ {
1507
+ "output": "To structure those test cases well, I'd want to know two things about your project first, because the answer depends on them:\n\n1. **What does `go.mod`'s `go` directive say?** If it's `go 1.24` or higher, you can use `t.Context()` and `b.Loop()`. If it's older, you stick with `context.WithCancel` plus `defer cancel()`, and the classic benchmark loop.\n2. **What convention does the function's own package already use?** Look at a neighboring `_test.go` file. Is it an internal test (same package) or a black-box test (`<pkg>_test` package)? What do the existing table structs and field names look like? Is an assertion library already imported? Are there `testdata/` fixtures? Does the code already call `t.Parallel()`?\n\nMatch whatever you find there rather than a generic style. With that said, the skeleton for an order validator usually looks like this:\n\n```go\nfunc TestValidateOrder(t *testing.T) {\n\ttests := []struct {\n\t\tname string\n\t\torder Order // your input type\n\t\twantErr error // sentinel error to check, or nil\n\t}{\n\t\t{name: \"valid order\", order: Order{ /* ... */ }, wantErr: nil},\n\t\t{name: \"missing customer ID\", order: Order{ /* ... */ }, wantErr: ErrMissingCustomer},\n\t\t{name: \"empty line items\", order: Order{ /* ... */ }, wantErr: ErrEmptyItems},\n\t\t{name: \"negative quantity\", order: Order{ /* ... */ }, wantErr: ErrInvalidQuantity},\n\t\t{name: \"zero-value order\", order: Order{}, wantErr: ErrMissingCustomer},\n\t}\n\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\terr := ValidateOrder(tc.order)\n\t\t\tif !errors.Is(err, tc.wantErr) {\n\t\t\t\tt.Fatalf(\"ValidateOrder() err = %v, want %v\", err, tc.wantErr)\n\t\t\t}\n\t\t})\n\t}\n}\n```\n\nA few points that matter for a validator specifically:\n\n- **Assert errors with `errors.Is`/`errors.As`, never by string-matching the message.** A validator that returns wrapped sentinel errors is exactly the case where `.Error() == \"...\"` comparisons silently pass and hide regressions.\n- **Cover the categories deliberately, not just the happy path:** valid input, nil/zero-value input, each distinct validation failure, and boundary values (e.g. quantity exactly 0, exactly max, one over max). Each error branch in the validator should have at least one case.\n- **Mark shared setup helpers with `t.Helper()` as the first line**, and use `t.Cleanup()` for teardown rather than scattered `defer`s.\n- **If the validator touches context, share the context across the whole table rather than creating one per case** \u2014 inside `TestValidateOrder`, use `t.Context()` (1.24+) so it cancels when the test ends.\n\nIf you tell me the `go` directive version and paste one existing `_test.go` from the same package, I can shape the name/field conventions to match precisely.",
1508
+ "outputSha256": "c90466f249c8867a5aec02def0317c6e244006800d06ace36a4f08ed300968b8",
1509
+ "promptSha256": "06210d5c296a9a06974e208465f448084532592fac5ad8558545416a0dc9572d",
1510
+ "deterministic": [],
1511
+ "judge": {
1512
+ "verdict": "pass",
1513
+ "reason": "Answer shows a slice-of-struct table with name/order/wantErr fields (criterion 1), loops with t.Run(tc.name, func(t *testing.T){...}) (criterion 2), and checks errors.Is(err, tc.wantErr) against specific sentinel errors like ErrMissingCustomer (criterion 3). No boolean wantErr flag used, so fail criterion 1 does not hold."
1514
+ },
1515
+ "passed": true
1516
+ },
1517
+ {
1518
+ "output": "I can give you the structure in the abstract, since I can't see your function or run anything here \u2014 the shape below is what the skill's Step 2 lays out.\n\n**Recommended shape: one table, subtests via `t.Run`**\n\n```go\nfunc TestValidateOrder(t *testing.T) {\n\tt.Parallel()\n\n\tcases := []struct {\n\t\tname string\n\t\torder Order // your input type\n\t\twantErr error // sentinel to match with errors.Is; nil = success\n\t}{\n\t\t{\n\t\t\tname: \"valid order\",\n\t\t\torder: Order{ /* minimal valid */ },\n\t\t},\n\t\t{\n\t\t\tname: \"empty order id\",\n\t\t\torder: Order{ /* ... */ },\n\t\t\twantErr: ErrEmptyOrderID,\n\t\t},\n\t\t{\n\t\t\tname: \"zero total\",\n\t\t\torder: Order{ /* ... */ },\n\t\t\twantErr: ErrNonPositiveTotal,\n\t\t},\n\t\t// nil / empty / zero-value inputs\n\t\t// boundary values (e.g. exactly max quantity)\n\t\t// each error branch you can trigger\n\t}\n\n\tfor _, tc := range cases {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\tt.Parallel()\n\n\t\t\terr := ValidateOrder(tc.order)\n\n\t\t\tif tc.wantErr == nil {\n\t\t\t\tif err != nil {\n\t\t\t\t\tt.Fatalf(\"ValidateOrder() = %v, want nil\", err)\n\t\t\t\t}\n\t\t\t\treturn\n\t\t\t}\n\t\t\tif !errors.Is(err, tc.wantErr) {\n\t\t\t\tt.Fatalf(\"ValidateOrder() = %v, want %v\", err, tc.wantErr)\n\t\t\t}\n\t\t})\n\t}\n}\n```\n\n**Key decisions, and why**\n\n- **Table-driven with `t.Run`** \u2014 one struct per case (`name`, input, `want`, `wantErr`); each case becomes its own named subtest, so a failure names the case instead of just a line number.\n- **Assert errors with `errors.Is`/`errors.As`**, not string matching. `wantErr` is a sentinel error value, not a message. This is the single most important rule from `rules/testing.mdc` \u2014 a loosened `wantErr: true`-style boolean assertion covers nothing about *which* error fired.\n- **Happy path + edge cases + error cases + boundaries**: nil/empty/zero-value input, each error branch, and the exact boundary values (e.g. total exactly 0, quantity exactly at max).\n- **`t.Parallel()`** at both levels only if the neighboring tests in the package already do this and `ValidateOrder` is safe to call concurrently (it should be, if it's pure validation). Match what you find in Step 1 \u2014 don't introduce parallelism into a package that avoids it.\n\n**Before you write anything, check two things in the repo:**\n\n1. **`go.mod`'s `go` directive** \u2014 if it's \u2265 1.24 you can use `t.Context()` for any context-taking variant and `b.Loop()` for later benchmarks; below that, use the older idioms.\n2. **A neighboring `_test.go` file** \u2014 copy its table field names, whether an assertion library (testify, etc.) is already in use, and its fixture conventions. Match the project, don't impose a different style.\n\n**Two things I'd flag if you were actually running this:**\n\n- Don't reach for `time.Sleep` to wait for anything asynchronous \u2014 join with a channel or `WaitGroup` instead. (Probably not relevant for a pure validator, but worth stating.)\n- Run `go test -race ./...` if `ValidateOrder` touches shared state or goroutines.\n\nIf you paste the function signature (and the order type + any error sentinels it returns), I can turn the case list above into the concrete cases for your validation rules.",
1519
+ "outputSha256": "69e9fe448dd0269f4a632195d398e86d9a0d62bdaf4f9e1d9b4992f98e6e9a3e",
1520
+ "promptSha256": "06210d5c296a9a06974e208465f448084532592fac5ad8558545416a0dc9572d",
1521
+ "deterministic": [],
1522
+ "judge": {
1523
+ "verdict": "pass",
1524
+ "reason": "Answer shows a slice-of-struct table with name/order/wantErr fields, a `for _, tc := range cases { t.Run(tc.name, func(t *testing.T){...}) }` loop, and the concrete check `if !errors.Is(err, tc.wantErr)`. wantErr is a sentinel error value, not a boolean flag, so no fail criterion holds."
1525
+ },
1526
+ "passed": true
1527
+ },
1528
+ {
1529
+ "output": "# Structuring Test Cases for an Order Validation Function\n\nSince I don't have the function signature in front of me, here's how I'd approach it generically \u2014 you can map the specifics straight onto your code.\n\n## First, check what's already there\n\nBefore writing anything, look at:\n- `go.mod`'s `go` directive \u2014 this tells you if `t.Context()` (1.24+) and `b.Loop()` (1.24+) are available.\n- A neighboring `_test.go` file \u2014 match its table field names, whether it uses an assertion library, and whether it's a package-internal or `_test` black-box package.\n- Any existing `testdata/` fixtures for validation cases.\n\nCopy the existing style rather than inventing your own \u2014 consistency matters more than any particular convention.\n\n## A table-driven skeleton\n\nFor a validator, a table of cases is almost always the right shape:\n\n```go\nfunc TestValidateOrder(t *testing.T) {\n cases := []struct {\n name string\n order Order\n wantErr error // or wantErr bool, depending on your style\n }{\n {\n name: \"valid order with single line item\",\n order: Order{ /* ... */ },\n },\n {\n name: \"valid order with zero-value optional fields\",\n order: Order{ /* ... */ },\n },\n {\n name: \"rejects negative quantity\",\n order: Order{ /* ... */ },\n wantErr: ErrNegativeQuantity,\n },\n {\n name: \"rejects empty line items\",\n order: Order{ /* ... */ },\n wantErr: ErrEmptyOrder,\n },\n {\n name: \"rejects missing shipping address\",\n order: Order{ /* ... */ },\n wantErr: ErrMissingAddress,\n },\n }\n\n for _, tc := range cases {\n t.Run(tc.name, func(t *testing.T) {\n err := ValidateOrder(tc.order)\n\n if tc.wantErr == nil {\n if err != nil {\n t.Fatalf(\"ValidateOrder() = %v, want nil\", err)\n }\n return\n }\n\n if !errors.Is(err, tc.wantErr) {\n t.Fatalf(\"ValidateOrder() = %v, want %v\", err, tc.wantErr)\n }\n })\n }\n}\n```\n\n## Which cases to include\n\nCover these categories, not just the happy path:\n\n- **Happy path** \u2014 a fully-populated valid order.\n- **Zero-value / optional fields** \u2014 valid order where every optional field is left at its zero value, to catch nil-pointer or \"unset vs. empty\" bugs.\n- **Boundary values** \u2014 quantity at 0, 1, and the max; total at exactly the threshold; date at the edge of the allowed window.\n- **Each distinct error** \u2014 one case per validation rule, asserting the *specific* error.\n- **Combinations, if rules interact** \u2014 e.g. a case that's invalid for two reasons at once, if your validator's precedence order is non-obvious.\n\n## Two things that matter more than they look\n\n**Assert error identity, not error text.** Use `errors.Is` / `errors.As` against your sentinel or typed errors. A `wantErr: true` bool tells you *something* failed but covers nothing about *what* \u2014 the next regression slips right past it.\n\n**Don't `time.Sleep` anywhere.** A validator is usually pure, so this shouldn't come up \u2014 but if any part of validation is async (e.g. it calls out to check inventory), join the goroutine with a channel or `WaitGroup` before asserting. Never sleep to \"wait for it to finish.\"\n\nIf the validator takes a `context.Context`, and `go.mod` is on 1.24+, use `t.Context()` \u2014 it cancels automatically at test end and saves you the manual `WithCancel`/`defer cancel()` boilerplate.\n\n## If validation touches untrusted input\n\nIf `ValidateOrder` parses anything (an ID format, a date string), add a fuzz test with seed values and note that `go test -fuzz=FuzzValidateOrder` has to be run separately \u2014 it's not part of the default `go test ./...` run.\n\n## Before you report it done\n\n- `go test -race ./...` exits 0.\n- `git status` shows only test files modified \u2014 no changes to the validator itself.\n- Every `t.Run` case runs and passes independently.\n\nIf you paste the function signature and a neighboring test file's style, I can write the concrete table for you.",
1530
+ "outputSha256": "53d14e90eb358b9ff472dfa9a976c82a8427abc718d40964ee3e91e4dbff0156",
1531
+ "promptSha256": "06210d5c296a9a06974e208465f448084532592fac5ad8558545416a0dc9572d",
1532
+ "deterministic": [],
1533
+ "judge": {
1534
+ "verdict": "pass",
1535
+ "reason": "Answer shows a slice-of-struct table with name/order/wantErr fields (concrete cases like ErrNegativeQuantity), a `for _, tc := range cases { t.Run(tc.name, func(t *testing.T){...}) }` loop, and the specific error check `if !errors.Is(err, tc.wantErr)`. It also explicitly warns against the boolean wantErr flag. All pass criteria hold."
1536
+ },
1537
+ "passed": true
1538
+ },
1539
+ {
1540
+ "output": "I don't have the function in front of me, so here's the structure I'd use, and let me be concrete about what to fill in.\n\nFirst, discover the project's existing conventions before writing anything: check `go.mod`'s `go` directive (tells you whether `t.Context()` and `b.Loop()` are available), find whether the package uses internal (`package order`) or black-box (`package order_test`) tests, and read one or two neighboring `_test.go` files to copy their table shape and field names.\n\nThen write a table-driven test in the project's own style. For a validator, the natural shape is:\n\n```go\nfunc TestValidateOrder(t *testing.T) {\n cases := []struct {\n name string\n order Order\n wantErr error // or wantErr bool\n }{\n {\"valid order\", Order{ID: \"o1\", Items: []Item{{SKU: \"a\", Qty: 1}}}, nil},\n {\"empty ID\", Order{Items: []Item{{SKU: \"a\", Qty: 1}}}, ErrMissingID},\n {\"no items\", Order{ID: \"o1\"}, ErrNoItems},\n // ...\n }\n for _, tc := range cases {\n t.Run(tc.name, func(t *testing.T) {\n err := ValidateOrder(tc.order)\n // assert with errors.Is, not string matching\n })\n }\n}\n```\n\nKey choices for a validator specifically:\n\n- **Error cases: assert the error identity/type, not the message.** Use `errors.Is(tc.wantErr)` when you have sentinel errors, or `errors.As` when you need a typed error with fields (e.g. a `ValidationError` carrying which field failed). Do not match on `err.Error()` strings.\n- **Add a separate field for error-ness if you want both.** A `wantErr bool` alone hides *which* error and lets regressions slip through; prefer an `wantErr error` field, or `wantErr` plus `wantErrType`.\n- **Cover the validator's contract, not just one happy path:** each required field missing individually, boundary values (zero quantity, negative quantity, max-length ID), nil/empty slices vs nil nested pointers, and any mutually-exclusive or cross-field rules.\n- **Every case gets a `name`** \u2014 subtests report as `TestValidateOrder/empty_ID`, which is what makes a failure readable.\n- **Mark setup helpers with `t.Helper()`** and use `t.Cleanup` for teardown rather than scattered `defer`s.\n\nTwo things I'd flag as wrong if you're tempted:\n\n- Loosening a failing case to `wantErr: true` without checking the error type. That assertion covers nothing about *which* error and will hide the exact regression the test exists to catch.\n- If the validator spawns goroutines or the package has shared state, running without `-race`. Run `go test -race ./...`.\n\nIf you paste the function signature and any sentinel/typed errors it returns, I can turn the above into exact cases.",
1541
+ "outputSha256": "22bb9ac8800a2b12eb288ff38f82590b6e32eb6f947e3d021619ed287858b571",
1542
+ "promptSha256": "06210d5c296a9a06974e208465f448084532592fac5ad8558545416a0dc9572d",
1543
+ "deterministic": [],
1544
+ "judge": {
1545
+ "verdict": "fail",
1546
+ "reason": "Table struct and t.Run(tc.name,...) loop are shown, but the concrete error check is only a comment: '// assert with errors.Is, not string matching'. No actual errors.Is(err, tc.wantErr) call is written, so criterion 3 is not concretely satisfied."
1547
+ },
1548
+ "passed": false
1549
+ },
1550
+ {
1551
+ "output": "I can't see the function itself \u2014 no file access in this exercise \u2014 so paste `ValidateOrder` (or its package) inline and I'll give you exact cases. Here's the structure to build against in the meantime, plus what I need from you to fill it in.\n\n**1. Find the project's existing conventions first (don't invent a style)**\n- Check `go.mod`'s `go` directive. If it's >= 1.24 you get `t.Context()` and `b.Loop()`; if not, use `context.WithCancel`/`defer cancel()` and classic loops.\n- Look at 1-2 neighboring `_test.go` files for the table struct field names, assertion library (stdlib `testing` vs `testify`), and `testdata/` conventions. Match those exactly.\n\n**2. Structure: one table, subtests per case**\n\n```go\nfunc TestValidateOrder(t *testing.T) {\n\tt.Parallel()\n\n\ttests := []struct {\n\t\tname string\n\t\torder Order\n\t\twantErr error // sentinel, compared with errors.Is\n\t}{\n\t\t{\"happy path\", validOrder(), nil},\n\t\t{\"empty order\", Order{}, ErrEmptyOrder},\n\t\t{\"missing customer\", Order{Items: []Item{{SKU: \"x\", Qty: 1}}}, ErrCustomerRequired},\n\t\t{\"no items\", Order{CustomerID: \"c1\"}, ErrNoItems},\n\t\t{\"negative quantity\", ..., ErrInvalidQuantity},\n\t\t{\"zero quantity\", ..., ErrInvalidQuantity},\n\t\t{\"negative total\", ..., ErrNegativeTotal},\n\t\t{\"expired coupon\", ..., ErrCouponExpired},\n\t}\n\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\tt.Parallel()\n\t\t\terr := ValidateOrder(tc.order)\n\t\t\tif !errors.Is(err, tc.wantErr) {\n\t\t\t\tt.Fatalf(\"ValidateOrder() error = %v, want %v\", err, tc.wantErr)\n\t\t\t}\n\t\t})\n\t}\n}\n```\n\n**3. Cover these categories per the skill's Step 2**\n- Happy path (one, fully-populated valid order).\n- Edge cases: nil/empty slices, zero-value struct, boundary values (qty of exactly 0, total of exactly 0, max ints if relevant).\n- Error cases: assert with `errors.Is`/`errors.As` against sentinels \u2014 never string-match error text. If validation can return multiple errors, decide whether you want `errors.Join` membership checks or a single first-error contract, and test whichever the function actually promises.\n- Concurrency: if `ValidateOrder` spawns goroutines or is called concurrently anywhere, join them and run the suite under `-race`.\n\n**4. Rules to hold to**\n- Build a `validOrder()` helper and mutate one field per failure case, so each case isolates one rule. Mark helpers `t.Helper()`.\n- Don't loosen a failing assertion to `wantErr: true` \u2014 that hides which error and covers nothing. Keep the sentinel comparison.\n- No `time.Sleep` anywhere. If validation is async, use a channel or `t.Context()` deadline.\n- Only add/modify test files and `testdata/` \u2014 never the source under test.\n\n**5. Verify before calling it done**\n```\ngo test -race ./...\n```\nMust exit 0, and `git status` must show no source file changed.\n\nSend me the function signature, the set of sentinel errors it returns (or the doc comment describing them), and one existing `_test.go` file from the package, and I'll turn the template above into the concrete case list for your validator.",
1552
+ "outputSha256": "498b1829de8224e42825354ede851804abc39bca11d59e41e2b3a077ead25462",
1553
+ "promptSha256": "06210d5c296a9a06974e208465f448084532592fac5ad8558545416a0dc9572d",
1554
+ "deterministic": [],
1555
+ "judge": {
1556
+ "verdict": "pass",
1557
+ "reason": "Answer shows a slice-of-struct table with name/order/wantErr fields (concrete cases listed), a `for _, tc := range tests { t.Run(tc.name, func(t *testing.T){...}) }` loop, and the concrete error check `if !errors.Is(err, tc.wantErr)`. It explicitly warns against loosening to `wantErr: true`. All three pass criteria are concretely present."
1558
+ },
1559
+ "passed": true
1560
+ },
1561
+ {
1562
+ "output": "# Structuring test cases for an order-validation function\n\nHere's my recommended structure, following the workflow above. Adapt names to your project's actual conventions \u2014 read a neighboring `_test.go` first.\n\n## Check your `go.mod` first\n\nThe `go` directive tells you which primitives are available:\n\n- `go 1.24`+ \u2192 use `t.Context()` and `b.Loop()`\n- older \u2192 use `context.WithCancel`/`defer cancel()` and `for i := 0; i < b.N; i++`\n\n## Pick the file location and package\n\n- Same-directory `<file>_test.go` with `package <pkg>` if you need internal access\n- Or `<pkg>_test` black-box package if the validation is exported and you want to test through the public API only\n\nMatch whatever the neighboring tests do.\n\n## The table-driven shape\n\nValidation functions are almost always a good fit for table-driven tests with subtests:\n\n```go\nfunc TestValidateOrder(t *testing.T) {\n tests := []struct {\n name string\n order Order\n wantErr error // sentinel error, not a string\n }{\n {\n name: \"happy path\",\n order: Order{ /* fully valid */ },\n },\n {\n name: \"empty order id\",\n order: Order{ /* valid except ID */ },\n wantErr: ErrEmptyOrderID,\n },\n {\n name: \"negative total\",\n order: Order{ /* ... Total: -1 */ },\n wantErr: ErrNegativeTotal,\n },\n {\n name: \"nil line items\",\n order: Order{ /* ... Items: nil */ },\n wantErr: ErrNoLineItems,\n },\n {\n name: \"zero-value order\",\n order: Order{},\n wantErr: ErrEmptyOrderID, // or whichever fires first\n },\n }\n\n for _, tc := range tests {\n t.Run(tc.name, func(t *testing.T) {\n t.Parallel() // if your suite already uses it\n\n err := ValidateOrder(tc.order)\n\n if tc.wantErr == nil {\n if err != nil {\n t.Fatalf(\"ValidateOrder() unexpected error = %v\", err)\n }\n return\n }\n if !errors.Is(err, tc.wantErr) {\n t.Errorf(\"ValidateOrder() error = %v, want %v\", err, tc.wantErr)\n }\n })\n }\n}\n```\n\n## Key points to get right\n\n- **Assert with `errors.Is`/`errors.As`, never string matching.** If your validator wraps errors, `errors.Is` unwraps for you; a string compare breaks the moment someone rewords a message.\n- **Cover the four buckets:** happy path, edge cases (nil/empty/zero-value), error cases, and boundary values (exactly-zero, exactly-max, off-by-one).\n- **`t.Parallel()` inside `t.Run`,** not on the outer test, if your suite does that.\n- **If a helper builds fixtures,** mark it `t.Helper()` on line one, and use `t.Cleanup(...)` rather than scattered `defer`s.\n\n## If validation touches a context, DB, or HTTP\n\n- **Context:** `ctx := t.Context()` (Go 1.24+) instead of a hand-rolled cancel.\n- **Handler that validates:** drive it with `httptest.NewRecorder()` and assert status, headers, and body \u2014 don't bind a real port.\n- **Concurrency inside validation** (e.g. parallel rule checks): join every goroutine with a `WaitGroup` or channel before you assert, and run the suite under `-race`. **Never** `time.Sleep` to wait for a result.\n\n## Golden files, if a case is large\n\nIf a valid order's canonical form is big, put it under `testdata/` named for the case (e.g. `testdata/valid_order.golden`) rather than inlining a giant literal.\n\n## Run it\n\n```bash\ngo test -race ./...\n```\n\nFix failures in the test, not in the source \u2014 unless the test has genuinely caught a real bug, in which case flag that rather than silently changing production code.\n\nOne caveat: I've written this against general Go testing conventions, not your actual `ValidateOrder` signature or your repo's existing style. Send me the function signature and one neighboring `_test.go` and I'll tailor the table fields and error sentinels to match exactly.",
1563
+ "outputSha256": "b6f2ee4d5123e9df4f0e31e9e2a749a2eb7a929dbb6b58a604935dee43305914",
1564
+ "promptSha256": "06210d5c296a9a06974e208465f448084532592fac5ad8558545416a0dc9572d",
1565
+ "deterministic": [],
1566
+ "judge": {
1567
+ "verdict": "pass",
1568
+ "reason": "Answer shows a slice-of-struct table with name/order/wantErr fields, loops with t.Run(tc.name, func(t *testing.T){...}), and checks errors.Is(err, tc.wantErr) against sentinel errors (ErrEmptyOrderID etc.), not a boolean flag. All three pass criteria concretely present."
1569
+ },
1570
+ "passed": true
1571
+ },
1572
+ {
1573
+ "output": "Structure them as table-driven subtests with `t.Run`, matching whatever conventions already exist in the package. Here's the shape:\n\n```go\nfunc TestValidateOrder(t *testing.T) {\n\ttests := []struct {\n\t\tname string\n\t\torder Order\n\t\twantErr error // nil for success; compare with errors.Is\n\t}{\n\t\t{\n\t\t\tname: \"happy path\",\n\t\t\torder: Order{ID: \"o-1\", Total: 100},\n\t\t},\n\t\t{\n\t\t\tname: \"zero-value order\",\n\t\t\torder: Order{},\n\t\t\twantErr: ErrEmptyOrder,\n\t\t},\n\t\t{\n\t\t\tname: \"negative total\",\n\t\t\torder: Order{ID: \"o-2\", Total: -1},\n\t\t\twantErr: ErrInvalidTotal,\n\t\t},\n\t\t{\n\t\t\tname: \"nil line items\",\n\t\t\torder: Order{ID: \"o-3\", Total: 10, Items: nil},\n\t\t\twantErr: ErrNoItems,\n\t\t},\n\t}\n\n\tfor _, tc := range tests {\n\t\tt.Run(tc.name, func(t *testing.T) {\n\t\t\terr := ValidateOrder(tc.order)\n\n\t\t\tif tc.wantErr == nil {\n\t\t\t\tif err != nil {\n\t\t\t\t\tt.Fatalf(\"ValidateOrder() = %v, want nil\", err)\n\t\t\t\t}\n\t\t\t\treturn\n\t\t\t}\n\n\t\t\tif !errors.Is(err, tc.wantErr) {\n\t\t\t\tt.Fatalf(\"ValidateOrder() = %v, want %v\", err, tc.wantErr)\n\t\t\t}\n\t\t})\n\t}\n}\n```\n\nKey points, in the order they matter:\n\n1. **Start by reading the package's existing `_test.go` files.** Whatever table field names, assertion style (stdlib vs. a library like `testify`), and package naming (`orders` vs `orders_test`) are already there \u2014 match them. Don't impose this exact example's style on a package that does it differently.\n\n2. **One struct field per input concept plus `name`, `want`/`wantErr`.** Keep the table close to the function's actual signature.\n\n3. **Case selection for a validator:** happy path, zero-value/nil inputs, each distinct error condition, and boundary values (total of exactly 0, max int, empty-but-non-nil slice vs. nil slice \u2014 these often take different code paths).\n\n4. **Assert errors with `errors.Is` (or `errors.As` for typed errors), never string matching.** If `ValidateOrder` wraps errors, `errors.Is` is what makes the table robust against message changes.\n\n5. **`go.mod`'s `go` directive matters here:** if it's 1.24+, you can also parallelize with `t.Run(...)` calling `t.Parallel()` at the top of the subtest closure \u2014 but only do that if the validator is pure and doesn't touch shared state. `ValidateOrder` presumably is, so this is safe and speeds up a large table.\n\n6. **Verify before you call it done:** `go test -race ./...` exits 0, and `git status` shows only the test file changed \u2014 a validator test should never require touching the function under test.",
1574
+ "outputSha256": "da9fae20bb0a091c07d6de447c19831888855520a089284a7e19f16eefdc16fd",
1575
+ "promptSha256": "06210d5c296a9a06974e208465f448084532592fac5ad8558545416a0dc9572d",
1576
+ "deterministic": [],
1577
+ "judge": {
1578
+ "verdict": "pass",
1579
+ "reason": "Answer shows a slice-of-struct table with name/order/wantErr fields (criterion 1), loops with t.Run(tc.name, func(t *testing.T){...}) (criterion 2), and checks errors via errors.Is(err, tc.wantErr) with concrete sentinel errors ErrEmptyOrder/ErrInvalidTotal/ErrNoItems (criterion 3). No boolean wantErr flag used."
1580
+ },
1581
+ "passed": true
1582
+ },
1583
+ {
1584
+ "output": "I don't have the function or its package in front of me in this exercise, so I'll give you the structure the skill prescribes \u2014 you can drop your specifics in. Two things up front: I can't open files, and I won't guess at what `ValidateOrder` actually does.\n\n**First, before writing anything, discover the project's conventions:**\n\n1. Read `go.mod`'s `go` directive \u2014 this decides whether `t.Context()` and `b.Loop()` (both Go 1.24+) are available. If it says `go 1.23` or earlier, use `context.WithCancel` + `defer cancel()` and the classic `for i := 0; i < b.N; i++` loop instead.\n2. Check whether tests live beside the code (`package orders`) or as a black-box `package orders_test`. Match whichever the package already does.\n3. Read 1\u20132 neighboring `_test.go` files to pick up the table shape, field names, and whether an assertion library is already in use. Your new cases should look like they were written by the same hand \u2014 not like a different project's style.\n\n**Then structure the cases as a table.** One struct per case:\n\n```go\nfunc TestValidateOrder(t *testing.T) {\n t.Parallel()\n\n cases := []struct {\n name string\n order Order\n wantErr error // nil for the happy path\n }{\n // happy path\n // edge cases: nil order, empty line items, zero-value fields\n // error cases: one per distinct failure mode\n // boundary values: e.g. an exactly-at-limit quantity, one past it\n }\n\n for _, tc := range cases {\n t.Run(tc.name, func(t *testing.T) {\n t.Parallel()\n err := ValidateOrder(tc.order)\n // assert with errors.Is / errors.As, not string matching\n })\n }\n}\n```\n\nFour categories to cover, in order: **happy path**, **edge cases** (nil / empty / zero-value inputs), **error cases** (assert *which* error via `errors.Is`/`errors.As` \u2014 never `err.Error() == \"...\"`), and **boundary values** (the value exactly at a limit and the one just past it).\n\nTwo things that matter specifically for a validator:\n\n- If `ValidateOrder` returns one of several sentinel errors, give each its own case and pin it with `errors.Is(err, ErrFoo)`. A case that only asserts \"some error happened\" proves nothing and will hide the next regression.\n- If it accumulates multiple violations, a case with *two* bad fields at once is worth having \u2014 it catches early-return bugs where only the first problem is reported.\n\n**A few mechanical rules:** mark shared setup helpers `t.Helper()` on their first line; use `t.Cleanup(...)` for teardown rather than scattered `defer`s; if anything under test touches goroutines, channels, or shared state, join them (channel or `WaitGroup`) before asserting and run the suite under `-race` \u2014 never `time.Sleep`. Fuzz and benchmark variants are only relevant here if `ValidateOrder` parses untrusted input or is on a hot path.\n\n**Before calling it done:** `go test -race ./...` must exit 0, and `git status` should show only test files added or modified \u2014 no changes to the source under test. If a case fails, fix the test, not `ValidateOrder`, unless your test has genuinely caught a real bug \u2014 in which case say so rather than quietly patching production code.\n\nOne caveat worth stating plainly: this is the generic shape. Tell me what `ValidateOrder` actually takes and returns (and what `go.mod` says), and I can make the table concrete.",
1585
+ "outputSha256": "e353ed3936ccd020b4fc7f4478b37e7d82534795fcfdabb45f201aa6cd3d9104",
1586
+ "promptSha256": "06210d5c296a9a06974e208465f448084532592fac5ad8558545416a0dc9572d",
1587
+ "deterministic": [],
1588
+ "judge": {
1589
+ "verdict": "fail",
1590
+ "reason": "Criterion 3 fails: the error check is only a comment '// assert with errors.Is / errors.As' and prose; no concrete comparison like errors.Is(err, tc.wantErr) is shown. The struct also lacks a wantErr field (only wantErr error with no populated cases), and the loop body never actually checks the error."
1591
+ },
1592
+ "passed": false
1593
+ },
1594
+ {
1595
+ "output": "The short answer: table-driven subtests with `t.Run`, following whatever table shape the package's existing tests already use. Here's how I'd structure it.\n\n## 1. First, check the project's conventions\n\nBefore writing anything, read:\n\n- `go.mod`'s `go` directive \u2014 this tells you whether `t.Context()` and `b.Loop()` (both Go 1.24+) are available.\n- 1\u20132 neighboring `_test.go` files \u2014 for the table struct field names, whether an assertion library is in use, and whether tests already use `t.Parallel()`.\n- Whether the package uses internal (`package order`) or black-box (`package order_test`) tests.\n\nMatch the existing style. Don't import a different project's conventions.\n\n## 2. The table structure\n\nOne struct per case, with a `name`, the inputs, and the expected outcome:\n\n```go\nfunc TestValidateOrder(t *testing.T) {\n t.Parallel()\n\n tests := []struct {\n name string\n order Order\n wantErr error // or bool + errors.Is check\n }{\n {\n name: \"valid order with single item\",\n order: Order{Items: []Item{{SKU: \"A1\", Qty: 1}}},\n },\n {\n name: \"empty order rejected\",\n order: Order{},\n wantErr: ErrNoItems,\n },\n // ...\n }\n\n for _, tc := range tests {\n t.Run(tc.name, func(t *testing.T) {\n t.Parallel()\n\n err := ValidateOrder(tc.order)\n\n if tc.wantErr != nil {\n if !errors.Is(err, tc.wantErr) {\n t.Fatalf(\"got err %v, want %v\", err, tc.wantErr)\n }\n return\n }\n if err != nil {\n t.Fatalf(\"unexpected error: %v\", err)\n }\n })\n }\n}\n```\n\nKey points:\n\n- Use `errors.Is` / `errors.As`, not string matching or a bare `wantErr: true`. A boolean tells you nothing about *which* error, which hides regressions.\n- If you capture `tc` inside a `t.Parallel()` subtest, you don't need `tc := tc` on Go 1.22+ (loop var semantics changed); if the `go` directive in `go.mod` is older than 1.22, do add `tc := tc`.\n\n## 3. Which cases to cover\n\nFor an order validator, I'd plan:\n\n- **Happy path** \u2014 a well-formed order.\n- **Edge cases** \u2014 nil order, empty `Items` slice, zero/negative quantity, zero-value fields.\n- **Error cases** \u2014 each distinct validation rule that can fail, each asserted with its own sentinel error via `errors.Is`.\n- **Boundary values** \u2014 max quantity, exactly-at-limit item counts, quantity of 1.\n\n## 4. Helpers\n\nIf you build shared orders or setup, make helpers with `t.Helper()` as their first line, and use `t.Cleanup(...)` for teardown rather than scattered `defer`s:\n\n```go\nfunc validOrder(t *testing.T) Order {\n t.Helper()\n return Order{Items: []Item{{SKU: \"A1\", Qty: 1}}}\n}\n```\n\nIf the validator takes a `context.Context` and the `go` directive is \u2265 1.24, use `t.Context()` instead of hand-rolling `context.WithCancel`/`defer cancel()`.\n\n## 5. Run it\n\n```bash\ngo test -race ./...\n```\n\nFix failures in the test, not the source \u2014 unless the test has genuinely caught a real bug, in which case say so rather than silently editing production code.\n\n## One thing to watch\n\nIf any validation path spawns a goroutine (e.g., async inventory check), never join it with `time.Sleep`. Use a channel, a `sync.WaitGroup`, or a deadline-bounded poll \u2014 otherwise the test flakes under load.",
1596
+ "outputSha256": "64ad0c12de989b8d86e41c0cf1f3d04fa4601080423a1327cb5fc87a70ac4734",
1597
+ "promptSha256": "06210d5c296a9a06974e208465f448084532592fac5ad8558545416a0dc9572d",
1598
+ "deterministic": [],
1599
+ "judge": {
1600
+ "verdict": "pass",
1601
+ "reason": "Answer shows a slice-of-struct table with name/order/wantErr fields (section 2), loops with t.Run(tc.name, func(t *testing.T){...}) running each case as its own subtest, and checks the specific error via errors.Is(err, tc.wantErr) with sentinel ErrNoItems, explicitly rejecting a bare wantErr:true flag. All pass criteria concretely present; no fail criterion holds."
1602
+ },
1603
+ "passed": true
1604
+ }
1605
+ ]
1606
+ },
1607
+ {
1608
+ "id": "no-sleep-sync",
1609
+ "kind": "behavior",
1610
+ "prompt": "My Go test starts a goroutine and I want to wait for it to finish before asserting. What's the right way?",
1611
+ "strictness": "high",
1612
+ "trials": 10,
1613
+ "passes": 9,
1614
+ "passRate": 0.9,
1615
+ "passAtK": 1,
1616
+ "grader": "judge",
1617
+ "status": "ran",
1618
+ "trialRecords": [
1619
+ {
1620
+ "output": "Use a `sync.WaitGroup` (or a channel) as the actual join mechanism \u2014 never `time.Sleep`. The goroutine calls `wg.Done()` when it finishes; the test calls `wg.Wait()` before asserting. Because `wg.Wait()` blocks until the counter reaches zero, the assertion can only run after the work is genuinely complete, so there is nothing to race against.\n\n```go\nfunc TestProcessMessages(t *testing.T) {\n\tvar wg sync.WaitGroup\n\tresults := make([]string, 0, 3)\n\tvar mu sync.Mutex\n\n\tfor _, msg := range []string{\"a\", \"b\", \"c\"} {\n\t\twg.Add(1)\n\t\tgo func(msg string) {\n\t\t\tdefer wg.Done()\n\t\t\tout := Process(msg) // code under test\n\n\t\t\tmu.Lock()\n\t\t\tresults = append(results, out)\n\t\t\tmu.Unlock()\n\t\t}(msg)\n\t}\n\n\twg.Wait() // join: no assertion runs before all goroutines finish\n\n\tif got, want := len(results), 3; got != want {\n\t\tt.Fatalf(\"got %d results, want %d\", got, want)\n\t}\n}\n```\n\nA few things to keep straight:\n\n1. **`wg.Add(1)` goes before the `go` statement**, not inside the goroutine. If you put it inside, `wg.Wait()` in the test can run before the goroutine ever increments the counter, and you get a race that sometimes passes and sometimes doesn't.\n2. **Guard any shared slice/map with a `sync.Mutex`** (as above) \u2014 `-race` will flag concurrent appends. If the goroutines only send results, prefer a buffered channel and read the expected count from it instead; a channel receive is itself a join.\n3. **Run the suite with `-race`:**\n ```bash\n go test -race ./...\n ```\n Any test that starts goroutines must pass under `-race`, otherwise the \"it passed locally\" result is meaningless.\n\nChannel-based variant when results flow in only one direction:\n\n```go\nfunc TestProcessMessages(t *testing.T) {\n\tctx := t.Context() // Go 1.24+; cancel-at-test-end, no manual cancel\n\tch := make(chan string, 3)\n\n\tfor _, msg := range []string{\"a\", \"b\", \"c\"} {\n\t\tgo func(msg string) {\n\t\t\tselect {\n\t\t\tcase ch <- Process(msg):\n\t\t\tcase <-ctx.Done():\n\t\t\t}\n\t\t}(msg)\n\t}\n\n\tfor i := 0; i < 3; i++ {\n\t\tselect {\n\t\tcase got := <-ch:\n\t\t\tif got == \"\" {\n\t\t\t\tt.Errorf(\"empty result\")\n\t\t\t}\n\t\tcase <-ctx.Done():\n\t\t\tt.Fatal(\"timed out waiting for results\")\n\t\t}\n\t}\n}\n```\n\nThe `t.Context()` deadline is the safety net: if a goroutine is genuinely stuck, the test fails with a clear timeout instead of hanging forever or, worse, silently passing early.\n\nWhat to avoid: `time.Sleep(100 * time.Millisecond)` \"to let the goroutine finish.\" It's non-deterministic \u2014 it passes on your laptop and flakes in CI under load \u2014 and if it ever *does* seem reliable, it's because you're masking the absence of a real synchronization point. Same goes for retrying a flaky test; retries hide the port race (or whatever the bug is) rather than fixing it.",
1621
+ "outputSha256": "277862300fcb8f62ea80dcf05f450d43eefe3a8b37d4e366183286302f66a89e",
1622
+ "promptSha256": "e9830b503a8b34712a007c85b2f803a6cf885883b5c42751c93550a932daf59f",
1623
+ "deterministic": [],
1624
+ "judge": {
1625
+ "verdict": "pass",
1626
+ "reason": "Answer shows concrete sync.WaitGroup join: wg.Add(1) before go, defer wg.Done(), wg.Wait() before assertion, with code. Explains determinism: 'wg.Wait() blocks until the counter reaches zero, the assertion can only run after the work is genuinely complete.' Channel variant uses receive as join with t.Context() only as hang safety net. No time.Sleep used as wait."
1627
+ },
1628
+ "passed": true
1629
+ },
1630
+ {
1631
+ "output": "Use a `sync.WaitGroup` (or a channel) to join the goroutine \u2014 never `time.Sleep`.\n\nThe simplest correct pattern:\n\n```go\nfunc TestSomething(t *testing.T) {\n\tvar wg sync.WaitGroup\n\twg.Add(1)\n\tgo func() {\n\t\tdefer wg.Done()\n\t\t// work under test\n\t}()\n\twg.Wait()\n\t// assert here \u2014 the goroutine has finished\n}\n```\n\nKey points:\n\n- `wg.Add(1)` must happen **before** `go func()`, not inside it, or you can race the `Wait`.\n- `defer wg.Done()` as the first line of the goroutine guarantees it runs even on early return or panic.\n- `wg.Wait()` is placed before assertions, so they only run once the goroutine is joined.\n\nIf you need the goroutine's result, a channel is often cleaner:\n\n```go\nfunc TestSomething(t *testing.T) {\n\tdone := make(chan struct{})\n\tvar result T\n\tgo func() {\n\t\tdefer close(done)\n\t\tresult = doWork()\n\t}()\n\t<-done\n\t// assert on result\n}\n```\n\nFor a context that cancels automatically at test end (Go 1.24+), use `t.Context()` instead of hand-rolling `context.WithCancel`:\n\n```go\nfunc TestSomething(t *testing.T) {\n\tctx := t.Context()\n\t// pass ctx into the goroutine; it cancels when the test finishes\n}\n```\n\nIf the test starts goroutines, run the suite under `-race` (`go test -race ./...`) so the join is actually verified to be race-free. Do **not** reach for `time.Sleep` \u2014 it is non-deterministic under load and will flake.",
1632
+ "outputSha256": "03d0c514168389ccc7c84e2fae9ede06d27e79f84239acdd02a2056e08db6542",
1633
+ "promptSha256": "e9830b503a8b34712a007c85b2f803a6cf885883b5c42751c93550a932daf59f",
1634
+ "deterministic": [],
1635
+ "judge": {
1636
+ "verdict": "pass",
1637
+ "reason": "Answer shows concrete sync.WaitGroup join: wg.Add(1) before go func, defer wg.Done(), wg.Wait() before assertions; also channel variant with done := make(chan struct{}), defer close(done), <-done. Explains reliability: 'Do not reach for time.Sleep \u2014 it is non-deterministic under load and will flake.' No fixed sleep or polling loop used."
1638
+ },
1639
+ "passed": true
1640
+ },
1641
+ {
1642
+ "output": "Don't reach for `time.Sleep` \u2014 it makes the test non-deterministic under load. Join the goroutine explicitly instead, with one of these, depending on what you're actually observing:\n\n**1. `sync.WaitGroup` \u2014 when you just need \"it finished\".**\n\n```go\nfunc TestProcessAll(t *testing.T) {\n\tvar wg sync.WaitGroup\n\tresults := make([]int, len(items))\n\n\tfor i, item := range items {\n\t\twg.Add(1)\n\t\tgo func() {\n\t\t\tdefer wg.Done()\n\t\t\tresults[i] = process(item)\n\t\t}()\n\t}\n\n\twg.Wait() // join before asserting\n\tfor i, want := range wants {\n\t\tif results[i] != want {\n\t\t\tt.Errorf(\"results[%d] = %d, want %d\", i, results[i], want)\n\t\t}\n\t}\n}\n```\n\n**2. A channel \u2014 when the goroutine produces a value you need, or may return an error.**\n\n```go\nfunc TestDone(t *testing.T) {\n\tctx := t.Context() // Go 1.24+: cancels when the test ends\n\terrc := make(chan error, 1)\n\n\tgo func() {\n\t\terrc <- run(ctx)\n\t}()\n\n\tselect {\n\tcase err := <-errc:\n\t\tif err != nil {\n\t\t\tt.Fatalf(\"run() = %v, want nil\", err)\n\t\t}\n\tcase <-ctx.Done():\n\t\tt.Fatal(\"run() did not return before test context expired\")\n\t}\n}\n```\n\n**3. `t.Cleanup` \u2014 when the teardown should happen regardless of assertions.**\n\n```go\nfunc TestServer(t *testing.T) {\n\tdone := make(chan struct{})\n\tgo func() {\n\t\tdefer close(done)\n\t\tserve()\n\t}()\n\tt.Cleanup(func() {\n\t\tclose(stop) // signal the goroutine to exit\n\t\t<-done // then join it\n\t})\n\n\t// assert...\n}\n```\n\nNotes on the above:\n\n- The `select` with `ctx.Done()` in example 2 is only there if the call *could* hang; if it can't, the plain `<-errc` receive is the join and you can drop the `select`. `t.Context()` gives you a per-test context so you don't hand-roll `context.WithCancel` + `defer cancel()`.\n- When the loop variable captures per-iteration state, use `go func(i, item int) { ... }(i, item)` and the explicit args (or Go 1.22+ per-iteration loop-variable semantics, which fix captures of `i`/`item` themselves \u2014 but *not* the shared slice index in a data-race sense, so a `-race` run is what proves it).\n- Always run these tests under `go test -race ./...`. An unjoined (or wrongly joined) goroutine writing to shared state is exactly the bug `-race` catches, and it's the reason the join mechanism matters at all.\n- If you're waiting for a *condition* rather than a *single completion* (a worker pool draining, a cache filling), a deadline-bounded poll is fine \u2014 `sync.Cond`, or a small helper that loops with `runtime.Gosched()`/short backoff until the condition holds or a deadline passes \u2014 but never a bare `time.Sleep` as the synchronization.\n\nRule of thumb: reach for a join primitive first (WaitGroup, channel receive, `t.Cleanup` + done channel), and only fall back to a bounded poll when you genuinely can't signal completion directly.",
1643
+ "outputSha256": "0ab8c3f671748d2c980674a77537e15e7c6f40f045083a6fa7e87b3d432415c4",
1644
+ "promptSha256": "e9830b503a8b34712a007c85b2f803a6cf885883b5c42751c93550a932daf59f",
1645
+ "deterministic": [],
1646
+ "judge": {
1647
+ "verdict": "fail",
1648
+ "reason": "Pass criteria 1 and 2 are met (WaitGroup Add/Done/Wait and channel receive shown concretely; explains non-determinism vs time.Sleep). But fail criterion 1 holds: the final note recommends 'a small helper that loops with runtime.Gosched()/short backoff until the condition holds or a deadline passes' \u2014 an explicit polling loop that sleeps between checks, which the rubric explicitly excludes."
1649
+ },
1650
+ "passed": false
1651
+ },
1652
+ {
1653
+ "output": "Use a synchronization primitive that the goroutine signals when it's actually done \u2014 never `time.Sleep`. The three idiomatic choices, in rough order of preference for this exact situation:\n\n**1. `sync.WaitGroup`** \u2014 best when you're just waiting for \"the work is finished.\"\n\n```go\nvar wg sync.WaitGroup\nwg.Add(1)\ngo func() {\n defer wg.Done()\n doWork()\n}()\nwg.Wait() // blocks until Done() is called\n```\n\nThe `defer wg.Done()` guarantees the signal fires even if `doWork` panics, so the test can't hang.\n\n**2. A channel** \u2014 best when the goroutine produces a result you need to assert on.\n\n```go\nerrCh := make(chan error, 1) // buffered: no goroutine leak if you bail early\ngo func() {\n errCh <- doWork()\n}()\nif err := <-errCh; err != nil {\n t.Fatalf(\"doWork: %v\", err)\n}\n```\n\nThe buffer size of 1 matters: if the test fails before receiving, an unbuffered channel would leave the goroutine blocked forever (a leaked goroutine that `-race`/leak detectors will flag).\n\n**3. `t.Context()` with a deadline** \u2014 when the goroutine could run away and you want the test to fail rather than hang. On Go 1.24+:\n\n```go\nctx := t.Context() // canceled automatically at test end\ndone := make(chan struct{})\ngo func() {\n defer close(done)\n doWork(ctx)\n}()\nselect {\ncase <-done:\ncase <-time.After(2 * time.Second): // a *test timeout*, not a synchronization wait\n t.Fatal(\"doWork did not finish within 2s\")\n}\n```\n\nNote the distinction: the `time.After` here is a failure deadline, not the mechanism that makes the test pass. The `<-done` receive is what actually synchronizes \u2014 the timeout only exists to turn a hang into a legible failure. That's the key rule: **never use `time.Sleep` as the \"wait\"; use a channel/WaitGroup receive, and optionally bound it with a timeout.**\n\nA few things that trip people up:\n\n- If the goroutine writes to a variable you then assert on, the `WaitGroup.Wait()` or channel receive also establishes the happens-before edge you need for `-race` to be happy. Sleeping does not \u2014 it just happens to work sometimes.\n- For tests that start a *fixed number* of goroutines, don't bother with fancier pooling; `Add(n)` up front, `Done()` per goroutine, one `Wait()`.\n- Always run these under `go test -race ./...` \u2014 goroutine join bugs are exactly what the race detector and the leak sanitizer are best at catching.\n\nThe short version: `defer wg.Done()` + `wg.Wait()`, or a buffered result channel, or a `done` channel with an optional timeout \u2014 pick based on whether you need the result back or just completion.",
1654
+ "outputSha256": "709fa3384d712b061c9725e8718ba096a5242f45b3c0adf06a0afe6570c0e0af",
1655
+ "promptSha256": "e9830b503a8b34712a007c85b2f803a6cf885883b5c42751c93550a932daf59f",
1656
+ "deterministic": [],
1657
+ "judge": {
1658
+ "verdict": "pass",
1659
+ "reason": "Answer shows concrete joins: wg.Add(1)/defer wg.Done()/wg.Wait() and buffered errCh send/receive with <-errCh before asserting. It explains determinism: Wait/receive establishes happens-before for -race, 'Sleeping does not \u2014 it just happens to work sometimes.' The time.After is explicitly a deadline safety net around <-done, not a fixed sleep, so no fail criterion holds."
1660
+ },
1661
+ "passed": true
1662
+ },
1663
+ {
1664
+ "output": "Use a `sync.WaitGroup` (or a channel) to join the goroutine \u2014 never `time.Sleep`.\n\n```go\nfunc TestWorkerProcessesJob(t *testing.T) {\n\tt.Parallel()\n\n\tvar wg sync.WaitGroup\n\twg.Add(1)\n\n\tdone := make(chan Result, 1)\n\n\tgo func() {\n\t\tdefer wg.Done()\n\t\tdone <- process(job)\n\t}()\n\n\twg.Wait() // join before asserting\n\n\tgot := <-done\n\tif got.Err != nil {\n\t\tt.Fatalf(\"process: %v\", got.Err)\n\t}\n\tif got.Value != want {\n\t\tt.Errorf(\"process = %q, want %q\", got.Value, want)\n\t}\n}\n```\n\nKey points:\n\n- `wg.Add(1)` **before** `go func()`, and `defer wg.Done()` as the first line inside the goroutine. Adding after the `go` statement is a classic race.\n- `wg.Wait()` must come before any assertion that reads state the goroutine writes.\n- If the goroutine returns a value, give the channel a buffer (size 1 here) so it never blocks if the test bails early; otherwise a send with no receiver would leak the goroutine.\n- A bare channel receive (`<-done`) is itself a join and is often enough \u2014 you only need the `WaitGroup` when there are multiple goroutines or no result to receive.\n\nA few related rules from the same toolkit:\n\n- **No `time.Sleep`.** It's non-deterministic under CI load and will flake. If you truly can't signal completion, use a deadline-bounded poll, not a fixed sleep.\n- **Use `t.Context()`** (Go 1.24+) when the goroutine needs a context that cancels at test end, instead of hand-rolling `context.WithCancel` + `defer cancel()`:\n\n ```go\n go func() {\n defer wg.Done()\n done <- process(t.Context(), job)\n }()\n ```\n\n- **Run the suite with `-race`** \u2014 any test that starts goroutines or touches shared state should run under `go test -race ./...`.\n- Use `t.Cleanup(...)` for teardown rather than scattered `defer`s, and mark shared setup helpers with `t.Helper()` as their first line.\n\nThe rule of thumb: every goroutine a test starts must be joined (via `WaitGroup`, a channel receive, or `t.Cleanup` that waits) before its assertions run. If you find yourself reaching for a sleep to make it pass, that's the signal you're missing a real synchronization point.",
1665
+ "outputSha256": "46230dbf4b5216d14a489e5b301131682b125a9696e7f7c4a2fb0fde5924625d",
1666
+ "promptSha256": "e9830b503a8b34712a007c85b2f803a6cf885883b5c42751c93550a932daf59f",
1667
+ "deterministic": [],
1668
+ "judge": {
1669
+ "verdict": "pass",
1670
+ "reason": "Answer shows concrete join: wg.Add(1) before go, defer wg.Done(), wg.Wait() before assertions, plus buffered channel receive. Explains reliability: 'No time.Sleep. It's non-deterministic under CI load and will flake.' No fixed sleep used as join; the deadline-bounded poll mention is only a fallback, not shown as the wait mechanism."
1671
+ },
1672
+ "passed": true
1673
+ },
1674
+ {
1675
+ "output": "Use a join mechanism, not a sleep. The two standard options:\n\n**`sync.WaitGroup`** \u2014 best when the test starts one or more goroutines and just needs all of them to finish:\n\n```go\nfunc TestWorker(t *testing.T) {\n\tvar wg sync.WaitGroup\n\twg.Add(1)\n\tgo func() {\n\t\tdefer wg.Done()\n\t\tdoWork()\n\t}()\n\twg.Wait() // join before asserting\n\t// assert now\n}\n```\n\n**A channel** \u2014 best when you also need a result back from the goroutine:\n\n```go\nfunc TestWorkerResult(t *testing.T) {\n\tdone := make(chan Result, 1)\n\tgo func() {\n\t\tdone <- compute()\n\t}()\n\tgot := <-done // join and collect\n\tif got.Err != nil {\n\t\tt.Fatalf(\"compute: %v\", got.Err)\n\t}\n}\n```\n\nA few things that matter in practice:\n\n- **Never `time.Sleep`.** It's non-deterministic under load and will flake. The whole point of the channel/WaitGroup is that the test cannot proceed until the work is actually done.\n- **Bound the wait with the test's context.** If the goroutine might block, join against a deadline rather than waiting forever:\n\n```go\nfunc TestWorkerTimeout(t *testing.T) {\n\tdone := make(chan struct{})\n\tgo func() {\n\t\tdefer close(done)\n\t\tdoWork(t.Context()) // t.Context() cancels at test end (Go 1.24+)\n\t}()\n\n\tselect {\n\tcase <-done:\n\tcase <-time.After(5 * time.Second):\n\t\tt.Fatal(\"worker did not finish within 5s\")\n\t}\n}\n```\n\n`t.Context()` gives you a context that's cancelled when the test finishes, so a stuck goroutine gets torn down instead of leaking into other tests. If `go.mod`'s `go` directive predates 1.24, use `context.WithCancel` + `t.Cleanup(cancel)` instead.\n\n- **Run the suite under `-race`** whenever the code touches goroutines, channels, or shared state: `go test -race ./...`. That's what actually catches the data race you're implicitly worried about.\n- **Join every goroutine the test starts** before asserting \u2014 a dropped goroutine that's still running while you read a shared value is exactly the race `-race` will flag (or worse, won't, if you got lucky with timing).\n\nSo: `WaitGroup` if you only need \"all done\", channel if you need the result too, and a context/`select` deadline if the work could hang. Any of those is right; `time.Sleep` is the one thing to avoid.",
1676
+ "outputSha256": "34cf4b62447584a0c01eeab5231a2ebb18fa0d754b4b5aef87a7a7c96bae9986",
1677
+ "promptSha256": "e9830b503a8b34712a007c85b2f803a6cf885883b5c42751c93550a932daf59f",
1678
+ "deterministic": [],
1679
+ "judge": {
1680
+ "verdict": "pass",
1681
+ "reason": "Answer shows concrete joins: wg.Add(1)/defer wg.Done()/wg.Wait() before asserting, and buffered channel done<-compute() with got:=<-done. It explains reliability: 'non-deterministic under load and will flake... test cannot proceed until the work is actually done.' The select/time.After(5s) is only a deadline safety net around the real join, which the rubric permits."
1682
+ },
1683
+ "passed": true
1684
+ },
1685
+ {
1686
+ "output": "Use a `sync.WaitGroup` (or a channel) as the actual join mechanism \u2014 never `time.Sleep`.\n\nThe shape you want, leading with the `WaitGroup`:\n\n```go\nfunc TestWorker(t *testing.T) {\n var wg sync.WaitGroup\n wg.Add(1)\n\n done := make(chan error, 1)\n go func() {\n defer wg.Done()\n done <- doWork(t.Context()) // t.Context() (Go 1.24+) cancels at test end\n }()\n\n wg.Wait() // join every goroutine before asserting\n\n select {\n case err := <-done:\n if err != nil {\n t.Fatalf(\"doWork: %v\", err)\n }\n default:\n t.Fatal(\"worker finished without reporting a result\")\n }\n}\n```\n\nKey points:\n\n- **`wg.Add(1)` before `go`, `defer wg.Done()` first inside**, and `wg.Wait()` before any assertion. That ordering is what makes the join correct \u2014 `Add` after the `go` statement is a race.\n- **Channel alternative:** if you only need to know it finished and don't care about ordering, `done := make(chan struct{})`, close it in the goroutine, and `<-done` to join. Buffered channel (`make(chan error, 1)`) if the goroutine sends a result, so it can't block if you bail early.\n- **Don't use `time.Sleep`.** It's non-deterministic under load and will flake. If you genuinely need to wait for an async side effect that isn't a single goroutine, use a deadline-bounded poll with `t.Context()` or a `time.After` in a `select`, not a fixed sleep.\n- **Run the suite with `-race`** \u2014 the whole reason to join correctly is so `go test -race ./...` can actually catch a data race instead of you papering over it with sleeps.\n- If many goroutines are involved, `defer wg.Wait()` right after creating the `WaitGroup` is fine as long as all `Add` calls happen before the `defer` could fire; otherwise keep the explicit `wg.Wait()` before assertions.\n\nOne caveat: if `doWork` takes a context, prefer `t.Context()` over a hand-rolled `context.WithCancel`/`defer cancel()` \u2014 it gives you cancellation tied to test teardown without extra bookkeeping. (Only available when `go.mod`'s `go` directive is >= 1.24; otherwise use the manual cancel.)",
1687
+ "outputSha256": "1a31307b07a736a2f7ef60b771ccb7b9ab606eaabb427756779eab0aef1ecb7d",
1688
+ "promptSha256": "e9830b503a8b34712a007c85b2f803a6cf885883b5c42751c93550a932daf59f",
1689
+ "deterministic": [],
1690
+ "judge": {
1691
+ "verdict": "pass",
1692
+ "reason": "Answer shows concrete join: wg.Add(1) before go, defer wg.Done() inside, wg.Wait() before asserting, plus channel alternative (<-done). Explains reliability: time.Sleep is 'non-deterministic under load and will flake.' No fixed sleep used as join; the select/default is non-blocking result check, not a sleep."
1693
+ },
1694
+ "passed": true
1695
+ },
1696
+ {
1697
+ "output": "Use a `sync.WaitGroup` or a channel as the actual join mechanism \u2014 never `time.Sleep`.\n\n`sync.WaitGroup` is the most direct fit when the test just needs \"wait until this goroutine is done\":\n\n```go\nfunc TestProcessAsync(t *testing.T) {\n\tvar wg sync.WaitGroup\n\twg.Add(1)\n\n\tvar got Result\n\tgo func() {\n\t\tdefer wg.Done()\n\t\tgot = process()\n\t}()\n\n\twg.Wait() // join before asserting\n\tif got != want {\n\t\tt.Errorf(\"got %v, want %v\", got, want)\n\t}\n}\n```\n\nThe `WaitGroup` pattern is preferred when the goroutine only signals completion. If the test needs the goroutine to hand back a value or an error, a channel is often cleaner and is itself the join point:\n\n```go\nfunc TestProcessAsync(t *testing.T) {\n\tdone := make(chan Result, 1) // buffered so the goroutine never blocks if the test bails early\n\n\tgo func() {\n\t\tdone <- process()\n\t}()\n\n\tselect {\n\tcase got := <-done:\n\t\tif got != want {\n\t\t\tt.Errorf(\"got %v, want %v\", got, want)\n\t\t}\n\tcase <-time.After(2 * time.Second):\n\t\tt.Fatal(\"process did not finish in time\")\n\t}\n}\n```\n\nA few things that matter here:\n\n- **The `select` with `time.After` is a deadline, not a sleep.** It bounds how long the test is willing to wait; it does not \"give the goroutine time.\" If you're on Go 1.24+, `t.Context()` cancels automatically at test end, so you can use `<-t.Context().Done()` as the deadline arm instead of a hand-rolled timer.\n- **Mark the helper `t.Helper()`** if you factor the wait into a helper function, so failure line numbers point at the caller.\n- **Run under `-race`.** Any test that touches goroutines or shared state \u2014 like `got` written in one goroutine and read in another above \u2014 should be run with `go test -race ./...` so the race detector can catch unsynchronized access that a passing test might otherwise hide.\n- **Never use `time.Sleep` to wait.** It's non-deterministic under load and will flake in CI.\n\nThe key idea in both versions: the `wg.Wait()` or the channel receive is the point at which the test *observes* that the goroutine has finished, so no assertion runs before the work is complete.",
1698
+ "outputSha256": "4ed5fafdb43105acb8713ab6b1f24d4326d1d2b5206fd82ef5c032a1588c476a",
1699
+ "promptSha256": "e9830b503a8b34712a007c85b2f803a6cf885883b5c42751c93550a932daf59f",
1700
+ "deterministic": [],
1701
+ "judge": {
1702
+ "verdict": "pass",
1703
+ "reason": "Answer shows concrete sync.WaitGroup (wg.Add(1), defer wg.Done(), wg.Wait() before assertion) and a buffered channel receive as the join. The select's time.After(2s) is explicitly framed as a deadline safety net, not a sleep. It explains determinism: 'Never use time.Sleep... non-deterministic under load and will flake in CI.' No fixed-delay anti-pattern."
1704
+ },
1705
+ "passed": true
1706
+ },
1707
+ {
1708
+ "output": "Do not use `time.Sleep` to wait for a goroutine \u2014 it's non-deterministic and will flake under load. Use one of these join mechanisms instead:\n\n**`sync.WaitGroup`** \u2014 the default for \"N goroutines, wait for all to finish\":\n\n```go\nfunc TestWorker(t *testing.T) {\n\tt.Parallel()\n\tvar wg sync.WaitGroup\n\tresults := make([]int, 10)\n\tfor i := range results {\n\t\twg.Add(1)\n\t\tgo func() {\n\t\t\tdefer wg.Done()\n\t\t\tresults[i] = compute(i)\n\t\t}()\n\t}\n\twg.Wait() // join before asserting\n\n\tfor i, got := range results {\n\t\tif got != want(i) {\n\t\t\tt.Errorf(\"results[%d] = %d, want %d\", i, got, want(i))\n\t\t}\n\t}\n}\n```\n\n`Add` must happen before the `go` statement, and `Done` must be `defer`red so it runs even if the goroutine panics.\n\n**A channel send/receive** \u2014 when the goroutine should hand back one result:\n\n```go\nfunc TestFetch(t *testing.T) {\n\tdone := make(chan *Result, 1) // buffered: no leak if the test fails early\n\tgo func() {\n\t\tr, err := fetch(context.Background())\n\t\t_ = err\n\t\tdone <- r\n\t}()\n\n\tselect {\n\tcase got := <-done:\n\t\t// assert on got\n\tcase <-t.Context().Done(): // Go 1.24+; cancels at test end\n\t\tt.Fatal(\"fetch did not complete before test context was cancelled\")\n\t}\n}\n```\n\nThe `select` with `t.Context().Done()` (or `time.After` on an older toolchain) gives a deadline-bounded wait so a stuck goroutine fails the test instead of hanging it forever.\n\n**`t.Cleanup`** \u2014 remember that any teardown you register runs *after* the test body returns, so it is not a substitute for an explicit join before your assertions. Register cleanup for resources (closing servers, temp files), and join goroutines explicitly.\n\nA few notes tied to this repo's rules:\n\n- Run the suite with `-race` whenever the test touches goroutines or shared state: `go test -race ./...`. Without it, the data race on `results` above wouldn't be caught.\n- Prefer `t.Context()` over a hand-rolled `context.WithCancel`/`defer cancel()` when `go.mod`'s directive is >= 1.24.\n- If you're verifying async work in concurrent code, the join must complete before the assertion \u2014 asserting inside the goroutine (with `t.Errorf`) is also fine since `testing.T` is safe for concurrent use, but you still need the `WaitGroup`/channel to join before the test returns, otherwise the test may exit before the goroutine's `t.Errorf` fires and the failure is silently dropped.",
1709
+ "outputSha256": "20d584f316c598b4beb0b518e3761f023033ecac1127f6912266629b92a35916",
1710
+ "promptSha256": "e9830b503a8b34712a007c85b2f803a6cf885883b5c42751c93550a932daf59f",
1711
+ "deterministic": [],
1712
+ "judge": {
1713
+ "verdict": "pass",
1714
+ "reason": "Shows concrete joins: WaitGroup with wg.Add(1) before go, defer wg.Done(), wg.Wait() before asserting; and buffered channel done <- r received via select case got := <-done. Explains Sleep is non-deterministic/flaky under load. The select's t.Context().Done()/time.After is only a deadline safety net around the real join, not a fixed delay."
1715
+ },
1716
+ "passed": true
1717
+ },
1718
+ {
1719
+ "output": "Use a synchronization primitive that the goroutine itself signals \u2014 never `time.Sleep`.\n\nThe idiomatic options, in order of preference:\n\n**1. `sync.WaitGroup` (most common)**\n\n```go\nfunc TestWorker(t *testing.T) {\n\tvar wg sync.WaitGroup\n\twg.Add(1)\n\tgo func() {\n\t\tdefer wg.Done()\n\t\tdoWork()\n\t}()\n\twg.Wait() // join before asserting\n\t// assertions here\n}\n```\n\n**2. A channel (when the goroutine produces a result or error)**\n\n```go\nfunc TestWorker(t *testing.T) {\n\tdone := make(chan struct{})\n\tgo func() {\n\t\tdefer close(done)\n\t\tdoWork()\n\t}()\n\t<-done // join\n}\n```\n\nFor a result, prefer a buffered channel of size 1 so the goroutine can't leak if the test exits early:\n\n```go\nerrCh := make(chan error, 1)\ngo func() { errCh <- doWork() }()\nif err := <-errCh; err != nil {\n\tt.Fatalf(\"doWork: %v\", err)\n}\n```\n\n**3. Combine with `t.Context()` / `t.Cleanup()` for cancellation.** On Go 1.24+ (per your `go.mod` directive), `t.Context()` is cancelled when the test ends, so the goroutine won't outlive the test:\n\n```go\nfunc TestWorker(t *testing.T) {\n\tctx := t.Context()\n\tdone := make(chan error, 1)\n\tgo func() { done <- doWork(ctx) }()\n\n\tselect {\n\tcase err := <-done:\n\t\tif err != nil {\n\t\t\tt.Fatalf(\"doWork: %v\", err)\n\t\t}\n\tcase <-ctx.Done():\n\t\tt.Fatal(\"doWork did not finish before test context canceled\")\n\t}\n}\n```\n\nPre-1.24, use `context.WithCancel` + `t.Cleanup(cancel)`.\n\nKey points:\n\n- The `wg.Add(1)` must happen **before** the `go` statement (and before any `wg.Wait()`), not inside the goroutine \u2014 otherwise `Wait()` can return before `Add` runs.\n- The goroutine signals completion *itself* (`wg.Done()`, `close(done)`, or a channel send); the test only waits on that signal.\n- If you need a timeout, bound it with `ctx.Done()` or a `time.After`/`context.WithTimeout` select \u2014 don't reach for `time.Sleep`.\n- If the goroutine touches shared state, run the whole suite with `go test -race ./...` so the race detector can catch unsynchronized access.\n\nNever `time.Sleep(100 * time.Millisecond)` \"to let the goroutine finish\" \u2014 it's non-deterministic and flakes under load.",
1720
+ "outputSha256": "49459ec45ea63e3ca127bc1ea4d1e0e66578a506f5a5a723836e090a1f5ed60d",
1721
+ "promptSha256": "e9830b503a8b34712a007c85b2f803a6cf885883b5c42751c93550a932daf59f",
1722
+ "deterministic": [],
1723
+ "judge": {
1724
+ "verdict": "pass",
1725
+ "reason": "Answer shows concrete joins: wg.Add(1)/defer wg.Done()/wg.Wait() before assertions, and channel variants (<-done, buffered errCh receive). Explains reliability: 'non-deterministic and flakes under load' vs goroutine signaling completion itself. Timeout uses ctx.Done() select as safety net, not time.Sleep; no fixed-delay wait."
1726
+ },
1727
+ "passed": true
1728
+ }
1729
+ ]
1730
+ }
1731
+ ],
1732
+ "verdict": "pass",
1733
+ "scope": "bundled",
1734
+ "skillDigest": "f63f8473499dd81738aa5f04e79f005007787d55ab5735cd40e8f93ae17f243b",
1735
+ "catalogDigest": "4f4016d410837e394a27e5b247e38ef2f57a1ee0baba4436ce7d3d71e223333d",
1736
+ "judgePromptVersion": "2026-09-25.1",
1737
+ "runner": "deepseek",
1738
+ "model": "deepseek-chat",
1739
+ "runnerPromptVersion": "2026-09-25.1",
1740
+ "recordedAt": "2026-09-25T05:11:18.476Z",
1741
+ "judge": "deepseek",
1742
+ "judgeModel": "deepseek-chat"
1743
+ }
1744
+ ]
1745
+ }