@mrciphersmith/keryx 0.3.3 → 0.3.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (140) hide show
  1. package/dist/cli.js +996 -366
  2. package/docs/README.md +2 -0
  3. package/package.json +1 -1
  4. package/src/gdskills/bundled/install-manifest.json +520 -48
  5. package/src/gdskills/bundled/stacks/csharp-dotnet/agent-refs.json +4 -0
  6. package/src/gdskills/bundled/stacks/csharp-dotnet/governance/eval.json +1881 -0
  7. package/src/gdskills/bundled/stacks/csharp-dotnet/governance/scout.json +33 -0
  8. package/src/gdskills/bundled/stacks/csharp-dotnet/pack.json +38 -0
  9. package/src/gdskills/bundled/stacks/csharp-dotnet/rules/coding-style.mdc +100 -0
  10. package/src/gdskills/bundled/stacks/csharp-dotnet/rules/patterns.mdc +107 -0
  11. package/src/gdskills/bundled/stacks/csharp-dotnet/rules/security.mdc +86 -0
  12. package/src/gdskills/bundled/stacks/csharp-dotnet/rules/testing.mdc +89 -0
  13. package/src/gdskills/bundled/stacks/csharp-dotnet/skills/dotnet-build-fix/SKILL.md +143 -0
  14. package/src/gdskills/bundled/stacks/csharp-dotnet/skills/dotnet-build-fix/evals.json +77 -0
  15. package/src/gdskills/bundled/stacks/csharp-dotnet/skills/dotnet-code-review/SKILL.md +121 -0
  16. package/src/gdskills/bundled/stacks/csharp-dotnet/skills/dotnet-code-review/evals.json +77 -0
  17. package/src/gdskills/bundled/stacks/csharp-dotnet/skills/dotnet-implementation/SKILL.md +134 -0
  18. package/src/gdskills/bundled/stacks/csharp-dotnet/skills/dotnet-implementation/evals.json +76 -0
  19. package/src/gdskills/bundled/stacks/csharp-dotnet/skills/dotnet-testing/SKILL.md +130 -0
  20. package/src/gdskills/bundled/stacks/csharp-dotnet/skills/dotnet-testing/evals.json +77 -0
  21. package/src/gdskills/bundled/stacks/django/agent-refs.json +3 -0
  22. package/src/gdskills/bundled/stacks/django/governance/eval.json +1763 -0
  23. package/src/gdskills/bundled/stacks/django/governance/scout.json +40 -0
  24. package/src/gdskills/bundled/stacks/django/pack.json +43 -0
  25. package/src/gdskills/bundled/stacks/django/rules/coding-style.mdc +80 -0
  26. package/src/gdskills/bundled/stacks/django/rules/patterns.mdc +92 -0
  27. package/src/gdskills/bundled/stacks/django/rules/security.mdc +92 -0
  28. package/src/gdskills/bundled/stacks/django/rules/testing.mdc +89 -0
  29. package/src/gdskills/bundled/stacks/django/skills/django-build-fix/SKILL.md +149 -0
  30. package/src/gdskills/bundled/stacks/django/skills/django-build-fix/evals.json +49 -0
  31. package/src/gdskills/bundled/stacks/django/skills/django-code-review/SKILL.md +137 -0
  32. package/src/gdskills/bundled/stacks/django/skills/django-code-review/evals.json +48 -0
  33. package/src/gdskills/bundled/stacks/django/skills/django-implementation/SKILL.md +147 -0
  34. package/src/gdskills/bundled/stacks/django/skills/django-implementation/evals.json +75 -0
  35. package/src/gdskills/bundled/stacks/django/skills/django-migrate/SKILL.md +166 -0
  36. package/src/gdskills/bundled/stacks/django/skills/django-migrate/evals.json +49 -0
  37. package/src/gdskills/bundled/stacks/django/skills/django-testing/SKILL.md +130 -0
  38. package/src/gdskills/bundled/stacks/django/skills/django-testing/evals.json +48 -0
  39. package/src/gdskills/bundled/stacks/fastapi/agent-refs.json +3 -0
  40. package/src/gdskills/bundled/stacks/fastapi/governance/eval.json +1777 -0
  41. package/src/gdskills/bundled/stacks/fastapi/governance/scout.json +34 -0
  42. package/src/gdskills/bundled/stacks/fastapi/pack.json +43 -0
  43. package/src/gdskills/bundled/stacks/fastapi/rules/coding-style.mdc +68 -0
  44. package/src/gdskills/bundled/stacks/fastapi/rules/patterns.mdc +108 -0
  45. package/src/gdskills/bundled/stacks/fastapi/rules/security.mdc +99 -0
  46. package/src/gdskills/bundled/stacks/fastapi/rules/testing.mdc +85 -0
  47. package/src/gdskills/bundled/stacks/fastapi/skills/fastapi-build-fix/SKILL.md +157 -0
  48. package/src/gdskills/bundled/stacks/fastapi/skills/fastapi-build-fix/evals.json +76 -0
  49. package/src/gdskills/bundled/stacks/fastapi/skills/fastapi-code-review/SKILL.md +150 -0
  50. package/src/gdskills/bundled/stacks/fastapi/skills/fastapi-code-review/evals.json +74 -0
  51. package/src/gdskills/bundled/stacks/fastapi/skills/fastapi-implementation/SKILL.md +158 -0
  52. package/src/gdskills/bundled/stacks/fastapi/skills/fastapi-implementation/evals.json +75 -0
  53. package/src/gdskills/bundled/stacks/fastapi/skills/fastapi-testing/SKILL.md +146 -0
  54. package/src/gdskills/bundled/stacks/fastapi/skills/fastapi-testing/evals.json +74 -0
  55. package/src/gdskills/bundled/stacks/flutter-dart/agent-refs.json +4 -0
  56. package/src/gdskills/bundled/stacks/flutter-dart/governance/eval.json +1849 -0
  57. package/src/gdskills/bundled/stacks/flutter-dart/governance/scout.json +33 -0
  58. package/src/gdskills/bundled/stacks/flutter-dart/pack.json +41 -0
  59. package/src/gdskills/bundled/stacks/flutter-dart/rules/coding-style.mdc +98 -0
  60. package/src/gdskills/bundled/stacks/flutter-dart/rules/patterns.mdc +88 -0
  61. package/src/gdskills/bundled/stacks/flutter-dart/rules/security.mdc +91 -0
  62. package/src/gdskills/bundled/stacks/flutter-dart/rules/testing.mdc +101 -0
  63. package/src/gdskills/bundled/stacks/flutter-dart/skills/flutter-build-fix/SKILL.md +134 -0
  64. package/src/gdskills/bundled/stacks/flutter-dart/skills/flutter-build-fix/evals.json +79 -0
  65. package/src/gdskills/bundled/stacks/flutter-dart/skills/flutter-code-review/SKILL.md +124 -0
  66. package/src/gdskills/bundled/stacks/flutter-dart/skills/flutter-code-review/evals.json +74 -0
  67. package/src/gdskills/bundled/stacks/flutter-dart/skills/flutter-implementation/SKILL.md +139 -0
  68. package/src/gdskills/bundled/stacks/flutter-dart/skills/flutter-implementation/evals.json +77 -0
  69. package/src/gdskills/bundled/stacks/flutter-dart/skills/flutter-testing/SKILL.md +134 -0
  70. package/src/gdskills/bundled/stacks/flutter-dart/skills/flutter-testing/evals.json +74 -0
  71. package/src/gdskills/bundled/stacks/java-kotlin-spring/agent-refs.json +3 -0
  72. package/src/gdskills/bundled/stacks/java-kotlin-spring/governance/eval.json +2194 -0
  73. package/src/gdskills/bundled/stacks/java-kotlin-spring/governance/scout.json +39 -0
  74. package/src/gdskills/bundled/stacks/java-kotlin-spring/pack.json +40 -0
  75. package/src/gdskills/bundled/stacks/java-kotlin-spring/rules/coding-style.mdc +67 -0
  76. package/src/gdskills/bundled/stacks/java-kotlin-spring/rules/patterns.mdc +65 -0
  77. package/src/gdskills/bundled/stacks/java-kotlin-spring/rules/security.mdc +69 -0
  78. package/src/gdskills/bundled/stacks/java-kotlin-spring/rules/testing.mdc +80 -0
  79. package/src/gdskills/bundled/stacks/java-kotlin-spring/skills/java-kotlin-spring-build-fix/SKILL.md +144 -0
  80. package/src/gdskills/bundled/stacks/java-kotlin-spring/skills/java-kotlin-spring-build-fix/evals.json +74 -0
  81. package/src/gdskills/bundled/stacks/java-kotlin-spring/skills/java-kotlin-spring-code-review/SKILL.md +129 -0
  82. package/src/gdskills/bundled/stacks/java-kotlin-spring/skills/java-kotlin-spring-code-review/evals.json +74 -0
  83. package/src/gdskills/bundled/stacks/java-kotlin-spring/skills/java-kotlin-spring-implementation/SKILL.md +147 -0
  84. package/src/gdskills/bundled/stacks/java-kotlin-spring/skills/java-kotlin-spring-implementation/evals.json +75 -0
  85. package/src/gdskills/bundled/stacks/java-kotlin-spring/skills/java-kotlin-spring-migrate/SKILL.md +139 -0
  86. package/src/gdskills/bundled/stacks/java-kotlin-spring/skills/java-kotlin-spring-migrate/evals.json +74 -0
  87. package/src/gdskills/bundled/stacks/java-kotlin-spring/skills/java-kotlin-spring-testing/SKILL.md +128 -0
  88. package/src/gdskills/bundled/stacks/java-kotlin-spring/skills/java-kotlin-spring-testing/evals.json +73 -0
  89. package/src/gdskills/bundled/stacks/kotlin-android/agent-refs.json +4 -0
  90. package/src/gdskills/bundled/stacks/kotlin-android/governance/eval.json +1889 -0
  91. package/src/gdskills/bundled/stacks/kotlin-android/governance/scout.json +34 -0
  92. package/src/gdskills/bundled/stacks/kotlin-android/pack.json +38 -0
  93. package/src/gdskills/bundled/stacks/kotlin-android/rules/coding-style.mdc +89 -0
  94. package/src/gdskills/bundled/stacks/kotlin-android/rules/patterns.mdc +96 -0
  95. package/src/gdskills/bundled/stacks/kotlin-android/rules/security.mdc +90 -0
  96. package/src/gdskills/bundled/stacks/kotlin-android/rules/testing.mdc +89 -0
  97. package/src/gdskills/bundled/stacks/kotlin-android/skills/compose-implementation/SKILL.md +150 -0
  98. package/src/gdskills/bundled/stacks/kotlin-android/skills/compose-implementation/evals.json +77 -0
  99. package/src/gdskills/bundled/stacks/kotlin-android/skills/kotlin-android-build-fix/SKILL.md +151 -0
  100. package/src/gdskills/bundled/stacks/kotlin-android/skills/kotlin-android-build-fix/evals.json +76 -0
  101. package/src/gdskills/bundled/stacks/kotlin-android/skills/kotlin-android-code-review/SKILL.md +139 -0
  102. package/src/gdskills/bundled/stacks/kotlin-android/skills/kotlin-android-code-review/evals.json +78 -0
  103. package/src/gdskills/bundled/stacks/kotlin-android/skills/kotlin-android-testing/SKILL.md +131 -0
  104. package/src/gdskills/bundled/stacks/kotlin-android/skills/kotlin-android-testing/evals.json +77 -0
  105. package/src/gdskills/bundled/stacks/python/agent-refs.json +2 -1
  106. package/src/gdskills/bundled/stacks/python/pack.json +1 -1
  107. package/src/gdskills/bundled/stacks/rust/agent-refs.json +3 -0
  108. package/src/gdskills/bundled/stacks/rust/governance/eval.json +1823 -0
  109. package/src/gdskills/bundled/stacks/rust/governance/scout.json +32 -0
  110. package/src/gdskills/bundled/stacks/rust/pack.json +42 -0
  111. package/src/gdskills/bundled/stacks/rust/rules/coding-style.mdc +93 -0
  112. package/src/gdskills/bundled/stacks/rust/rules/patterns.mdc +85 -0
  113. package/src/gdskills/bundled/stacks/rust/rules/security.mdc +85 -0
  114. package/src/gdskills/bundled/stacks/rust/rules/testing.mdc +82 -0
  115. package/src/gdskills/bundled/stacks/rust/skills/rust-build-fix/SKILL.md +141 -0
  116. package/src/gdskills/bundled/stacks/rust/skills/rust-build-fix/evals.json +78 -0
  117. package/src/gdskills/bundled/stacks/rust/skills/rust-code-review/SKILL.md +127 -0
  118. package/src/gdskills/bundled/stacks/rust/skills/rust-code-review/evals.json +72 -0
  119. package/src/gdskills/bundled/stacks/rust/skills/rust-implementation/SKILL.md +133 -0
  120. package/src/gdskills/bundled/stacks/rust/skills/rust-implementation/evals.json +79 -0
  121. package/src/gdskills/bundled/stacks/rust/skills/rust-testing/SKILL.md +130 -0
  122. package/src/gdskills/bundled/stacks/rust/skills/rust-testing/evals.json +75 -0
  123. package/src/gdskills/bundled/stacks/swift-ios/agent-refs.json +4 -0
  124. package/src/gdskills/bundled/stacks/swift-ios/governance/eval.json +1803 -0
  125. package/src/gdskills/bundled/stacks/swift-ios/governance/scout.json +32 -0
  126. package/src/gdskills/bundled/stacks/swift-ios/pack.json +38 -0
  127. package/src/gdskills/bundled/stacks/swift-ios/rules/coding-style.mdc +92 -0
  128. package/src/gdskills/bundled/stacks/swift-ios/rules/patterns.mdc +112 -0
  129. package/src/gdskills/bundled/stacks/swift-ios/rules/security.mdc +78 -0
  130. package/src/gdskills/bundled/stacks/swift-ios/rules/testing.mdc +90 -0
  131. package/src/gdskills/bundled/stacks/swift-ios/skills/swift-build-fix/SKILL.md +144 -0
  132. package/src/gdskills/bundled/stacks/swift-ios/skills/swift-build-fix/evals.json +75 -0
  133. package/src/gdskills/bundled/stacks/swift-ios/skills/swift-code-review/SKILL.md +122 -0
  134. package/src/gdskills/bundled/stacks/swift-ios/skills/swift-code-review/evals.json +75 -0
  135. package/src/gdskills/bundled/stacks/swift-ios/skills/swift-testing/SKILL.md +131 -0
  136. package/src/gdskills/bundled/stacks/swift-ios/skills/swift-testing/evals.json +75 -0
  137. package/src/gdskills/bundled/stacks/swift-ios/skills/swiftui-implementation/SKILL.md +149 -0
  138. package/src/gdskills/bundled/stacks/swift-ios/skills/swiftui-implementation/evals.json +76 -0
  139. package/src/gdskills/bundled/agents/python-build-fixer.md +0 -52
  140. package/src/gdskills/bundled/agents/python-code-auditor.md +0 -49
@@ -0,0 +1,1823 @@
1
+ {
2
+ "schemaVersion": "1.0.0",
3
+ "reports": [
4
+ {
5
+ "schemaVersion": "1.0.0",
6
+ "skillId": "rust/rust-build-fix",
7
+ "strictness": "high",
8
+ "trials": 10,
9
+ "triggerAccuracy": {
10
+ "truePositive": 4,
11
+ "falsePositive": 0,
12
+ "positives": 6,
13
+ "negatives": 7
14
+ },
15
+ "evidence": "authored",
16
+ "scenarios": [
17
+ {
18
+ "id": "trigger-positive-1",
19
+ "kind": "trigger-positive",
20
+ "prompt": "cargo build is failing with a trait bound not satisfied error",
21
+ "strictness": "high",
22
+ "trials": 1,
23
+ "passes": 1,
24
+ "passRate": 1,
25
+ "passAtK": 1,
26
+ "grader": "trigger-rank-fork-family",
27
+ "status": "ran",
28
+ "deterministic": true
29
+ },
30
+ {
31
+ "id": "trigger-positive-2",
32
+ "kind": "trigger-positive",
33
+ "prompt": "I bumped a dependency version and now cargo build complains the lockfile doesn't match Cargo.toml anymore, what do I do?",
34
+ "strictness": "high",
35
+ "trials": 1,
36
+ "passes": 0,
37
+ "passRate": 0,
38
+ "passAtK": 0,
39
+ "grader": "trigger-rank-fork-family",
40
+ "status": "ran",
41
+ "deterministic": true
42
+ },
43
+ {
44
+ "id": "trigger-positive-3",
45
+ "kind": "trigger-positive",
46
+ "prompt": "The compiler won't let me borrow `orders` as mutable twice at the same time in this function, how do I restructure it so it compiles?",
47
+ "strictness": "high",
48
+ "trials": 1,
49
+ "passes": 0,
50
+ "passRate": 0,
51
+ "passAtK": 0,
52
+ "grader": "trigger-rank-fork-family",
53
+ "status": "ran",
54
+ "deterministic": true
55
+ },
56
+ {
57
+ "id": "trigger-positive-4",
58
+ "kind": "trigger-positive",
59
+ "prompt": "cargo clippy is reporting a needless clone lint",
60
+ "strictness": "high",
61
+ "trials": 1,
62
+ "passes": 1,
63
+ "passRate": 1,
64
+ "passAtK": 1,
65
+ "grader": "trigger-rank-fork-family",
66
+ "status": "ran",
67
+ "deterministic": true
68
+ },
69
+ {
70
+ "id": "trigger-positive-5",
71
+ "kind": "trigger-positive",
72
+ "prompt": "This Rust code has a lifetime error that won't compile",
73
+ "strictness": "high",
74
+ "trials": 1,
75
+ "passes": 1,
76
+ "passRate": 1,
77
+ "passAtK": 1,
78
+ "grader": "trigger-rank-fork-family",
79
+ "status": "ran",
80
+ "deterministic": true
81
+ },
82
+ {
83
+ "id": "trigger-positive-6",
84
+ "kind": "trigger-positive",
85
+ "prompt": "cargo test is failing with a panic in the order crate, fix the build",
86
+ "strictness": "high",
87
+ "trials": 1,
88
+ "passes": 1,
89
+ "passRate": 1,
90
+ "passAtK": 1,
91
+ "grader": "trigger-rank-fork-family",
92
+ "status": "ran",
93
+ "deterministic": true
94
+ },
95
+ {
96
+ "id": "trigger-negative-1",
97
+ "kind": "trigger-negative",
98
+ "prompt": "npm install is failing with a peer dependency conflict",
99
+ "strictness": "high",
100
+ "trials": 1,
101
+ "passes": 1,
102
+ "passRate": 1,
103
+ "passAtK": 1,
104
+ "grader": "trigger-rank-fork-family",
105
+ "status": "ran",
106
+ "deterministic": true
107
+ },
108
+ {
109
+ "id": "trigger-negative-2",
110
+ "kind": "trigger-negative",
111
+ "prompt": "go build is failing for this Go module",
112
+ "strictness": "high",
113
+ "trials": 1,
114
+ "passes": 1,
115
+ "passRate": 1,
116
+ "passAtK": 1,
117
+ "grader": "trigger-rank-fork-family",
118
+ "status": "ran",
119
+ "deterministic": true
120
+ },
121
+ {
122
+ "id": "trigger-negative-3",
123
+ "kind": "trigger-negative",
124
+ "prompt": "pip install is failing for this Python project",
125
+ "strictness": "high",
126
+ "trials": 1,
127
+ "passes": 1,
128
+ "passRate": 1,
129
+ "passAtK": 1,
130
+ "grader": "trigger-rank-fork-family",
131
+ "status": "ran",
132
+ "deterministic": true
133
+ },
134
+ {
135
+ "id": "trigger-negative-4",
136
+ "kind": "trigger-negative",
137
+ "prompt": "Implement a new Rust feature in the order crate",
138
+ "strictness": "high",
139
+ "trials": 1,
140
+ "passes": 1,
141
+ "passRate": 1,
142
+ "passAtK": 1,
143
+ "grader": "trigger-rank-fork-family",
144
+ "status": "ran",
145
+ "deterministic": true
146
+ },
147
+ {
148
+ "id": "trigger-negative-5",
149
+ "kind": "trigger-negative",
150
+ "prompt": "Review this Rust diff for unsafe blocks and unwrap panics",
151
+ "strictness": "high",
152
+ "trials": 1,
153
+ "passes": 1,
154
+ "passRate": 1,
155
+ "passAtK": 1,
156
+ "grader": "trigger-rank-fork-family",
157
+ "status": "ran",
158
+ "deterministic": true
159
+ },
160
+ {
161
+ "id": "trigger-negative-6",
162
+ "kind": "trigger-negative",
163
+ "prompt": "Write Rust tests for this function",
164
+ "strictness": "high",
165
+ "trials": 1,
166
+ "passes": 1,
167
+ "passRate": 1,
168
+ "passAtK": 1,
169
+ "grader": "trigger-rank-fork-family",
170
+ "status": "ran",
171
+ "deterministic": true
172
+ },
173
+ {
174
+ "id": "trigger-negative-7",
175
+ "kind": "trigger-negative",
176
+ "prompt": "Our CMake configure step is failing with a missing header error, this is a C++ project not a Cargo one",
177
+ "strictness": "high",
178
+ "trials": 1,
179
+ "passes": 1,
180
+ "passRate": 1,
181
+ "passAtK": 1,
182
+ "grader": "trigger-rank-fork-family",
183
+ "status": "ran",
184
+ "deterministic": true
185
+ },
186
+ {
187
+ "id": "lockfile-mismatch",
188
+ "kind": "behavior",
189
+ "prompt": "cargo build --locked fails saying Cargo.lock is out of date after I bumped a dependency version in Cargo.toml (this is the same command our CI runs). How do I fix it?",
190
+ "strictness": "high",
191
+ "trials": 10,
192
+ "passes": 10,
193
+ "passRate": 1,
194
+ "passAtK": 1,
195
+ "grader": "judge",
196
+ "status": "ran",
197
+ "trialRecords": [
198
+ {
199
+ "output": "`--locked` means \"resolve nothing; fail if `Cargo.lock` doesn't already match `Cargo.toml`.\" Since you edited the version requirement, the lockfile no longer satisfies it, so cargo refuses. That's a real stale-lock condition, not a spurious CI failure — the fix is to re-resolve locally and commit the updated lockfile, not to change the CI command.\n\nDo this, in order:\n\n1. **Re-resolve the lockfile against your edited `Cargo.toml`:**\n ```bash\n cargo update -p <crate-name>\n ```\n Use the targeted form if only one dependency changed. `cargo update` with no `-p` re-resolves everything, which can drag in unrelated bumps you didn't intend — avoid it here since your change was scoped to one version bump.\n\n If the bump was a transitive dependency rather than a direct one, `cargo update -p <crate> --precise <version>` lets you pin exactly the version you want.\n\n2. **Inspect what actually changed** before committing:\n ```bash\n git diff Cargo.lock\n ```\n Confirm the diff only touches the crate(s) you bumped (plus any of its own transitive deps). If unrelated packages moved, that's a sign you ran a full `cargo update` — scope it back down.\n\n3. **Verify with the exact CI command, plus the full local gate:**\n ```bash\n cargo build --locked --all-targets\n cargo clippy --locked --all-targets -- -D warnings\n cargo fmt --check\n cargo test --locked\n ```\n `--locked` on the build is the one that matters for CI parity — if it passes locally, CI will too.\n\n4. **Commit both files together.** `Cargo.toml` and `Cargo.lock` must move in the same commit; a lockfile-only change or a manifest-only change is exactly what breaks `--locked` for everyone else.\n\nTwo things to watch for:\n\n- **If the version you bumped to doesn't exist or isn't reachable** (e.g. you typed a major that's not published, or a yanked release), `cargo update -p` will fail to resolve. Check the version exists on crates.io first, or use `--precise` to pick a published one.\n- **If the re-resolve forces other packages to move** because your bump pulled a newer shared transitive dep, that's expected and fine, but call it out in the commit message so reviewers know the lockfile churn is intentional and not an accidental full `cargo update`.\n\nDo not \"fix\" this by dropping `--locked` from CI, and do not hand-edit `Cargo.lock` — it's generated, and a hand-edited lockfile will fail the checksum validation the same way. Re-resolve with cargo, commit both files, verify with `--locked`.",
200
+ "outputSha256": "a9df773856c90f3fb4832e58e6c2f4243d393875a41e9171aef6c72bf1c83f04",
201
+ "promptSha256": "3f359775f43e4339282d923ee9ed7a6f0bcc9715d6ccdad422b655c568096d37",
202
+ "deterministic": [],
203
+ "judge": {
204
+ "verdict": "pass",
205
+ "reason": "Names `cargo update -p <crate-name>` (and `cargo update`) as the concrete fix, run from the crate root. Explains root cause: `--locked` fails because lockfile no longer satisfies edited Cargo.toml. Verification shown: `cargo build --locked --all-targets`, `cargo test --locked`. Warns against hand-editing Cargo.lock and dropping --locked; no edition/rust-version bump."
206
+ },
207
+ "passed": true
208
+ },
209
+ {
210
+ "output": "`--locked` is telling you the truth: CI is enforcing \"Cargo.toml and Cargo.lock must already agree,\" and they no longer do, because you edited Cargo.toml's version requirement without regenerating the lockfile. That is the entire bug — nothing in your source is wrong.\n\nThe one-line fix is to re-resolve the lockfile from the edited manifest:\n\n```bash\ncargo update -p <crate>\n```\n\nwhere `<crate>` is the dependency you bumped (targeted), or `cargo update` if the bump rippled across several crates and you're fine re-resolving the whole graph. That rewrites `Cargo.lock` to a state consistent with the new `Cargo.toml`, after which:\n\n```bash\ncargo build --locked # the CI command, now should pass\n```\n\nCommit the regenerated `Cargo.lock` alongside the `Cargo.toml` change. CI with `--locked` depends on the lockfile being committed, so a Cargo.toml edit without a matching lockfile commit is exactly the failure you hit.\n\nA few things worth knowing so this doesn't recur or get \"fixed\" the wrong way:\n\n- **Don't reach for `cargo build` (without `--locked`) to make CI's build pass locally.** That silently updates the lockfile in your working tree and makes the local build green while CI's `--locked` build is still red until you commit the updated lock. Use `cargo update` explicitly so the change is deliberate and visible in the diff.\n- **Check you didn't accidentally over-bump.** If `cargo update -p <crate>` can't satisfy the requirement in Cargo.toml, the version you wrote may not exist, or it conflicts with what another dependency needs. `cargo tree -i <crate>` shows who else depends on it before you loosen or force anything.\n- **Never delete Cargo.lock, and never let the fix be \"edit the requirement to whatever happens to resolve.\"** For an application/library binary the lockfile is meant to be pinned and committed; the right change is the smallest requirement edit that reflects the version you actually want, plus the regenerated lock.\n\nThen verify the full set:\n\n```bash\ncargo build --locked\ncargo clippy --all-targets -- -D warnings\ncargo fmt --check\ncargo test\n```\n\nReport it as: root cause — Cargo.lock was not regenerated after a Cargo.toml dependency version bump, so `--locked` (CI) correctly rejected the stale lock; fix — `cargo update -p <crate>` and commit the updated Cargo.lock.",
211
+ "outputSha256": "f994eb1eb067fae462367dc4b090d1380c250897eb932397eb019fbb526244c6",
212
+ "promptSha256": "3f359775f43e4339282d923ee9ed7a6f0bcc9715d6ccdad422b655c568096d37",
213
+ "deterministic": [],
214
+ "judge": {
215
+ "verdict": "pass",
216
+ "reason": "Names `cargo update -p <crate>` / `cargo update` as the fix (targeted or full re-resolve). States root cause: Cargo.toml version requirement edited without regenerating lockfile, so --locked rejects stale lock. Verification shown: `cargo build --locked`, plus clippy/fmt/test. Warns against hand-editing lockfile and against loosening requirement arbitrarily; checks `cargo tree -i <crate>` for conflicts. No fail criteria met."
217
+ },
218
+ "passed": true
219
+ },
220
+ {
221
+ "output": "`cargo build --locked` never updates `Cargo.lock` — it does exactly what CI wants: fail if the lockfile and `Cargo.toml` disagree instead of silently re-resolving. Bumping a dependency version in `Cargo.toml` always invalidates the lockfile until you re-resolve it, so the fix is to regenerate the lockfile and commit it, then re-run the locked build.\n\n**Do this:**\n\n1. Re-resolve just the crate you bumped, not the whole graph:\n ```bash\n cargo update -p <crate>\n ```\n If that complains (e.g. the new version needs a transitive dep that also has to move), fall back to a full `cargo update`. A targeted `-p` update keeps the diff small; a full update can touch unrelated crates, which you don't want in the same commit.\n\n2. Confirm the result is what you intended before trusting it:\n ```bash\n cargo tree -i <crate> # what depends on it, which version now resolves\n git diff Cargo.lock # should show only the expected version moves\n ```\n\n3. Verify exactly as CI will:\n ```bash\n cargo build --locked --all-targets\n cargo clippy --locked --all-targets -- -D warnings\n cargo fmt --check\n cargo test --locked\n ```\n All must exit 0. The `--locked` flag is the point here — a plain `cargo build` would re-resolve on the fly and hide whether the committed lockfile is actually in sync.\n\n4. Commit `Cargo.lock` together with the `Cargo.toml` bump. For a binary/workspace crate the lockfile is meant to be committed; if this is a library and `.gitignore` excludes `Cargo.lock`, then `--locked` in CI is the wrong command and you should reconsider the CI invocation rather than the lockfile.\n\n**What not to do:** don't edit `Cargo.lock` by hand, and don't drop `--locked` from the CI command just to get it green — that defeats the reproducibility guarantee `--locked` exists to enforce. If `cargo update -p <crate>` pulls a version you didn't want (a major bump, or something that changes your MSRV), that's a signal to pin a tighter requirement in `Cargo.toml` rather than to accept whatever resolved.\n\n**Root cause in one sentence:** bumping the version requirement in `Cargo.toml` invalidated `Cargo.lock`, and `--locked` correctly refused to re-resolve it — so regenerate the lockfile with `cargo update -p <crate>` and commit it alongside the manifest change.",
222
+ "outputSha256": "2e5226764c5721a6ce7aff4dc9cd696128382ebb914ceadb999b7a3abfbe6c41",
223
+ "promptSha256": "3f359775f43e4339282d923ee9ed7a6f0bcc9715d6ccdad422b655c568096d37",
224
+ "deterministic": [],
225
+ "judge": {
226
+ "verdict": "pass",
227
+ "reason": "Names `cargo update -p <crate>` (step 1) with full `cargo update` fallback, run from workspace root. States root cause: bumping Cargo.toml version invalidated Cargo.lock and --locked refuses to re-resolve. Verification: cargo build --locked --all-targets, cargo test --locked. Warns against hand-editing Cargo.lock. No edition/rust-version bump or blind dep bump."
228
+ },
229
+ "passed": true
230
+ },
231
+ {
232
+ "output": "Your CI runs `cargo build --locked`, which refuses to touch `Cargo.lock` at all — so once `Cargo.toml` asks for a version the lockfile doesn't contain, it fails rather than re-resolving. Fix it by re-resolving the lockfile locally, then commit the result.\n\n## What's actually happening\n\n`--locked` means \"build strictly from `Cargo.lock`; error instead of updating it.\" When you changed the version requirement in `Cargo.toml`, `Cargo.lock` no longer satisfies it. The command that repairs the lockfile is `cargo update` (or a targeted variant), run **without** `--locked`:\n\n```bash\n# Targeted: bump just the one crate you changed\ncargo update -p <crate-name>\n\n# If several deps moved or you're not sure which entry is stale\ncargo update\n```\n\n`cargo update -p <crate>` is almost always what you want here — it re-resolves only that package and leaves the rest of `Cargo.lock` alone, which keeps the diff minimal (this is the \"smallest change\" principle; a full `cargo update` can churn unrelated transitive versions).\n\nIf `cargo update -p <crate>` reports a version conflict, see what's pulling the conflicting version before bumping anything by hand:\n\n```bash\ncargo tree -i <crate>\n```\n\n## Verify it the way CI will\n\nReproduce CI's exact command, plus the rest of the gate:\n\n```bash\ncargo build --locked\ncargo clippy --all-targets -- -D warnings\ncargo fmt --check\ncargo test\n```\n\n`cargo build --locked` must now exit 0 without touching the lockfile. Commit both `Cargo.toml` and the regenerated `Cargo.lock` in the same change — a stale lockfile that only exists locally is exactly what caused the CI failure.\n\n## Report it\n\n```\nFixed: Cargo.lock was stale against a version bump in Cargo.toml\n - Root cause: changed the version requirement in Cargo.toml without re-resolving Cargo.lock,\n so `cargo build --locked` (CI) had no satisfying entry to build from\n - Ran `cargo update -p <crate>` to re-resolve just that package\n - cargo build --locked / clippy -D warnings / fmt --check / test all pass\n - Cargo.toml + Cargo.lock committed together\n```\n\n## Two things not to do\n\n- Don't drop `--locked` from the CI command to make it pass. That only hides the out-of-date lockfile for now and reintroduces the same class of failure on the next dependency change; the lockfile is meant to pin CI exactly.\n- Don't hand-edit `Cargo.lock` or bump the crate's `edition`/`rust-version` to dodge a conflict you haven't diagnosed. If `cargo update -p` surfaces a real version conflict, understand which dependency constrains it (`cargo tree -i <crate>`) before changing a version requirement to resolve it.",
233
+ "outputSha256": "2542a6295279097ef500a1ada19cd0019e832a9ef422aa16426cfa0aa79ce083",
234
+ "promptSha256": "3f359775f43e4339282d923ee9ed7a6f0bcc9715d6ccdad422b655c568096d37",
235
+ "deterministic": [],
236
+ "judge": {
237
+ "verdict": "pass",
238
+ "reason": "Names `cargo update -p <crate>` and `cargo update` run from workspace root (code block). States root cause: Cargo.lock no longer satisfies Cargo.toml after version bump. Verification: `cargo build --locked`, clippy, fmt, test. Checks `cargo tree -i <crate>` before hand-bumping. Warns against hand-editing Cargo.lock and edition/rust-version bumps."
239
+ },
240
+ "passed": true
241
+ },
242
+ {
243
+ "output": "`--locked` is doing exactly what it's documented to do: it tells Cargo \"fail rather than touch `Cargo.lock`,\" which is what you want in CI so a build never silently resolves different versions than what was committed. When you bump a dependency's version requirement in `Cargo.toml`, `Cargo.lock` no longer satisfies that requirement, so `--locked` errors out instead of updating it — that's the intended behavior, not a bug.\n\nThe fix is to regenerate `Cargo.lock` locally and commit it, rather than dropping `--locked` from CI (dropping it would let CI resolve whatever versions happen to exist at build time, which is the drift `--locked` exists to prevent).\n\n**Regenerate the lockfile:**\n\n```bash\n# targeted: only re-resolve the crate you bumped\ncargo update -p <crate> --precise <version>\n\n# or, if the bump is broader / you're not sure of the precise target\ncargo update\n```\n\n`cargo update -p <crate>` is usually the right choice here: it re-resolves just that package (and anything transitively forced to move) and leaves the rest of `Cargo.lock` untouched, so the CI diff stays minimal and reviewable. A bare `cargo update` re-resolves everything, which can pull in unrelated version bumps you didn't intend.\n\n**Then verify the exact command CI runs, plus the rest of the gate:**\n\n```bash\ncargo build --locked --all-targets\ncargo clippy --locked --all-targets -- -D warnings\ncargo fmt --check\ncargo test --locked\n```\n\nAll four must exit 0 — in particular re-run with `--locked` to confirm the regenerated lockfile now satisfies `Cargo.toml`, since that's the failing mode.\n\n**Commit both files together.** The `Cargo.toml` version bump and the regenerated `Cargo.lock` have to land in the same commit; committing the manifest change alone is what leaves CI (and every teammate) red until someone regenerates the lockfile.\n\n**One thing worth checking before you commit:** run `cargo tree -i <crate>` to see what else depends on the crate you bumped. If something else pins it to the old major version, `cargo update -p` may fail with a version conflict rather than silently working, and the real fix is reconciling the conflicting requirements in the manifests — not forcing the lockfile. If `cargo update -p` succeeds cleanly, the bump was compatible and you're done.\n\nI have no tools here, so I can't run any of this for you — the commands above are what to run in your own shell.",
244
+ "outputSha256": "bfb995b63fdcce2cdfffdb8d21ade796e9fe2b8a5f4e4416bc6b3686ec47f5f4",
245
+ "promptSha256": "3f359775f43e4339282d923ee9ed7a6f0bcc9715d6ccdad422b655c568096d37",
246
+ "deterministic": [],
247
+ "judge": {
248
+ "verdict": "pass",
249
+ "reason": "Names `cargo update -p <crate>` and `cargo update` as concrete commands (Regenerate the lockfile section). States root cause: Cargo.lock no longer satisfies Cargo.toml's bumped requirement. Verification: cargo build --locked --all-targets, cargo test --locked. Also checks cargo tree -i <crate> for conflicts. No edition/rust-version bump or hand-editing Cargo.lock."
250
+ },
251
+ "passed": true
252
+ },
253
+ {
254
+ "output": "`--locked` deliberately refuses to re-resolve dependencies: it tells cargo \"use exactly what's in `Cargo.lock`, error out if the manifest requires something the lockfile doesn't satisfy.\" Bumping a version in `Cargo.toml` makes the lockfile stale, so `--locked` fails by design. You fix it by refreshing the lockfile *before* the locked build, then committing the regenerated lockfile.\n\n## Fix\n\n1. **Regenerate the lockfile locally** (do this without `--locked`):\n\n ```bash\n cargo update -p <crate-name> # targeted: only re-resolve that one crate\n ```\n\n Use plain `cargo update` only if you actually need to re-resolve everything. `-p <crate>` keeps the diff minimal, which matters for reviewing what changed.\n\n2. **Check what actually moved** before trusting it:\n\n ```bash\n git diff Cargo.lock\n cargo tree -i <crate-name> # confirms the resolved version and who depends on it\n ```\n\n3. **Verify the exact CI command now passes:**\n\n ```bash\n cargo build --locked --all-targets\n cargo clippy --locked --all-targets -- -D warnings\n cargo test --locked\n ```\n\n4. **Commit the updated `Cargo.lock`** alongside the `Cargo.toml` change. This is the actual fix — CI's `--locked` will keep failing on every run until the regenerated lockfile is in the repo.\n\n## The key point\n\n`Cargo.lock` is a committed artifact. Changing `Cargo.toml`'s version requirements without updating `Cargo.lock` in the same commit leaves the two out of sync, and `--locked` exists specifically to catch that. So the fix is not to drop `--locked` from CI — it's to keep the lockfile in sync with the manifest.\n\n## A couple of caveats\n\n- If `cargo update -p <crate>` still can't satisfy the new requirement, the version you bumped to is likely in a genuine conflict with something else — check `cargo tree -i <crate>` to see what's pinning it. Don't force it by hand-editing `Cargo.lock` (cargo will just overwrite your edit, and a hand-edited lockfile is worse than a stale one).\n- If this is a library/binary whose `edition` or `rust-version` you also changed to make something compile, that's a separate decision from the lockfile sync — make sure it's intentional and not just an attempt to silence an unrelated error.\n- If CI installs with `--frozen` anywhere, that's `--locked` plus \"don't touch the network at all,\" so the regenerated lockfile must be committed, not generated during the CI run.\n\nAfter committing the refreshed `Cargo.lock`, `cargo build --locked` (and the rest of the CI pipeline) will pass again.",
255
+ "outputSha256": "a330818fc718ebb9b109f8bf47a59b22283688e9ea7fced1a546e055169a29a2",
256
+ "promptSha256": "3f359775f43e4339282d923ee9ed7a6f0bcc9715d6ccdad422b655c568096d37",
257
+ "deterministic": [],
258
+ "judge": {
259
+ "verdict": "pass",
260
+ "reason": "Names `cargo update -p <crate-name>` (and plain `cargo update`) as the fix, run locally before the locked build. Explains root cause: bumping Cargo.toml makes lockfile stale, `--locked` fails by design. Verification shown: `cargo build --locked --all-targets`, `cargo test --locked`. Warns against hand-editing Cargo.lock and against edition/rust-version changes to silence errors."
261
+ },
262
+ "passed": true
263
+ },
264
+ {
265
+ "output": "`--locked` is telling you exactly what it means: `Cargo.toml` now requires a version that `Cargo.lock` doesn't currently pin, so cargo refuses to touch the lock file (because `--locked` forbids that) and bails. CI uses `--locked` on purpose — it doesn't want the lock file mutated at build time, so it depends on you having committed an up-to-date `Cargo.lock` alongside the `Cargo.toml` change. Fix it locally, then commit the regenerated lock file.\n\n## Fix\n\nRe-resolve just the crate you bumped, rather than the whole graph:\n\n```bash\ncargo update -p <the-crate-you-bumped>\n```\n\nIf the version bump touched several crates or you want a full re-resolve:\n\n```bash\ncargo update\n```\n\nThen confirm it actually moved to a version satisfying the new `Cargo.toml` requirement:\n\n```bash\ncargo tree -i <the-crate-you-bumped>\n```\n\nThat shows the resolved version and what pulls it in. If `cargo update -p <crate>` can't find a satisfying version, the error will name the version conflict — then check `cargo tree -i <crate>` to see whether something else in the graph is pinning the old major, and resolve that constraint instead of forcing the bump.\n\nOnce the lock file is regenerated and committed, CI's `cargo build --locked` will pass because the lock file is now consistent with the manifest.\n\n## Verify the exact CI command\n\n```bash\ncargo build --locked\ncargo clippy --locked --all-targets -- -D warnings\ncargo fmt --check\ncargo test --locked\n```\n\nAll must exit 0, and `git status` should show `Cargo.lock` as a staged/committed change — the lock file update is part of the fix, not incidental.\n\n## Cautions that apply here specifically\n\n- **Don't \"fix\" this by dropping `--locked` from CI.** The flag is doing its job: it's catching a manifest/lock drift before it reaches a build. Removing it just lets CI silently re-resolve dependencies on every run, which is exactly the nondeterminism `--locked` exists to prevent.\n- **Don't hand-edit `Cargo.lock`.** Edit the manifest or run `cargo update`; the lock file's internal consistency (the `[[package]]` graph, version hashes) is maintained by cargo, not by hand.\n- **Watch for an accidental major bump.** If the version you changed in `Cargo.toml` crosses a semver-major boundary, `cargo update` may pull in a new major with breaking API changes; that's the same class of \"raised a dependency version to make an error disappear\" problem the skill warns about — make sure you know *why* the new version is needed, not just that it makes the manifest requirement satisfiable.\n- **If a transitive constraint blocks the new version**, resolve the constraint via `cargo tree -i <crate>` (what depends on the conflicting version) rather than pinning around it.\n\n## Report shape\n\n```\nFixed: Cargo.lock stale after a version bump in Cargo.toml\n - Root cause: Cargo.lock still pinned the old version after the manifest\n requirement changed\n - Ran `cargo update -p <crate>`; resolved version now <x.y.z>\n - cargo build --locked / clippy --locked -D warnings / fmt --check /\n test --locked all pass; Cargo.lock staged for commit\n```\n\nState the root cause in one sentence — \"Cargo.lock was stale against the manifest\" — not just \"the build passes now.\"",
266
+ "outputSha256": "aa501eb0c5f4cef83f83e27190d395cfcb0f798f6c0c199ca8bd182284f5c0a6",
267
+ "promptSha256": "3f359775f43e4339282d923ee9ed7a6f0bcc9715d6ccdad422b655c568096d37",
268
+ "deterministic": [],
269
+ "judge": {
270
+ "verdict": "pass",
271
+ "reason": "Names `cargo update -p <crate>` and `cargo update` as the fix (Fix section), states root cause: Cargo.toml requires a version Cargo.lock doesn't pin (opening + report shape), and gives verification via `cargo build --locked`/`cargo test --locked`. Warns against hand-editing Cargo.lock and against bumping another dep without checking cargo tree -i."
272
+ },
273
+ "passed": true
274
+ },
275
+ {
276
+ "output": "Your CI runs `cargo build --locked` (or the equivalent `--frozen`), which treats any drift between `Cargo.toml` and `Cargo.lock` as a hard error — it refuses to touch the lockfile at all, so a bump in `Cargo.toml` that hasn't been reflected in `Cargo.lock` fails immediately. That is exactly the situation you're in: the lockfile still records the old resolved version.\n\nThe fix is to re-resolve and commit the lockfile, not to touch `Cargo.toml` again. Two cases govern which command you use:\n\n**Case 1 — the bump is a version requirement you *changed* in `Cargo.toml` (e.g. you edited `foo = \"1.2\"` to `foo = \"1.5\"`).** Run the targeted update locally without `--locked`, so Cargo is allowed to rewrite `Cargo.lock`:\n\n```bash\ncargo update -p foo\n```\n\nThen build again with the exact CI command to confirm it's satisfied:\n\n```bash\ncargo build --locked\n```\n\nIf `cargo update -p foo` doesn't move it (the constraint still resolves to the old version, or another dependency pins it), check what's pulling the old version before reaching for anything broader:\n\n```bash\ncargo tree -i foo\n```\n\n**Case 2 — you added a brand-new dependency to `Cargo.toml` and the lockfile has no entry for it at all.** A targeted `-p` update can't help since the package isn't in the lockfile yet. Use a full re-resolve:\n\n```bash\ncargo update\ncargo build --locked\n```\n\n(`cargo update` with no `-p` re-resolves the whole graph and rewrites `Cargo.lock`; it still respects the version requirements already in `Cargo.toml`, so it won't silently jump majors on you.)\n\n**Commit the result.** `Cargo.lock` is the file your CI depends on — the change is only complete once the regenerated lockfile is committed alongside the `Cargo.toml` edit that caused it. CI's `--locked` then passes because the lockfile genuinely matches.\n\nTwo things not to do here:\n\n- Don't run `cargo update` with a version selector like\n `cargo update -p foo --precise 2.0.0` unless you specifically intend to override the requirement — that's a different action from reconciling the lockfile with what `Cargo.toml` already says, and it can push a version past the range a consumer expects. Understand why the resolved version changed before forcing it.\n- Don't \"fix\" it by editing `Cargo.lock` by hand, or by removing `--locked` from CI. The lockfile is machine-generated state; hand edits get clobbered on the next resolve and hide the real drift, and dropping `--locked` lets CI silently resolve something different from what you tested, which is the exact class of inconsistency `--locked` exists to prevent.\n\nIf after a proper `cargo update -p foo` (or full `cargo update`) the `--locked` build *still* reports the lockfile out of date, that means the lockfile wasn't actually rewritten — most often because it's staged/checked out read-only, the working tree isn't the one being built, or a `.cargo/config.toml` points elsewhere. Check `git status` shows `Cargo.lock` as modified and that you're building the same tree CI builds before concluding the update failed.",
277
+ "outputSha256": "97a5c87860c6e154ebbb565828f6e83cc3a8fb50e44ad53193e88d6106ec7e95",
278
+ "promptSha256": "3f359775f43e4339282d923ee9ed7a6f0bcc9715d6ccdad422b655c568096d37",
279
+ "deterministic": [],
280
+ "judge": {
281
+ "verdict": "pass",
282
+ "reason": "Names concrete commands `cargo update -p foo` and `cargo update` (Case 1/2) to regenerate the stale lockfile. States root cause: lockfile still records old resolved version, drift between Cargo.toml and Cargo.lock. Verification: `cargo build --locked` after update. Checks `cargo tree -i foo` before broader changes. Warns against hand-editing Cargo.lock and removing --locked. No fail criteria met."
283
+ },
284
+ "passed": true
285
+ },
286
+ {
287
+ "output": "The `--locked` flag is exactly what's failing you here: it tells Cargo \"resolve dependencies, but don't touch `Cargo.lock` — and error out if the lockfile would need to change.\" Bumping a version in `Cargo.toml` *always* changes what the lockfile needs to record, so `--locked` will refuse until the lockfile is regenerated. That regeneration is the fix, not a workaround.\n\n## Fix\n\nRun an update to regenerate the lockfile, with the same command shape your CI respects:\n\n```bash\ncargo update -p <crate-you-bumped>\n```\n\n`-p` targets just that one crate (and its transitive deps), which keeps the diff to `Cargo.lock` minimal and reviewable — preferable to a bare `cargo update`, which re-resolves everything and can drag in unrelated bumps.\n\nIf the version constraint you set is an exact `=x.y.z` and the registry doesn't yet have a matching entry, Cargo will say so here; that's a real signal to check whether the version exists rather than something to force past.\n\n## Verify\n\n```bash\ncargo build --locked --all-targets\ncargo clippy --all-targets -- -D warnings\ncargo fmt --check\ncargo test\n```\n\nThe `--locked` build now passing is the proof the lockfile is back in sync — that's precisely the command CI runs, so a clean local run means CI's resolve step will agree.\n\n## Commit\n\nCommit `Cargo.toml` **and** `Cargo.lock` together in the same commit. The failure came from them being committed out of step; splitting them across commits reproduces the same break for anyone building an intermediate revision.\n\n## Don't\n\n- Don't delete `Cargo.lock` and let it regenerate from scratch in CI — a full re-resolve can silently float every other dependency to newer versions, turning a one-crate bump into a broad, unreviewed change.\n- Don't drop `--locked` from the CI command to make it pass — that removes the guarantee the lockfile is committed at all, which is the check's whole purpose.\n- Don't hand-edit `Cargo.lock`. It's machine-generated; regenerate it.\n\nRoot cause in one sentence: bumping the version in `Cargo.toml` made `Cargo.lock` stale, and `--locked` correctly refuses to build a lockfile that no longer matches the manifest — running `cargo update -p <crate>` regenerates it, and both files get committed together.",
288
+ "outputSha256": "3580fd4eda17b4e6ee10a5530dc9a88282a0e143110451af1ea3158b3f790735",
289
+ "promptSha256": "3f359775f43e4339282d923ee9ed7a6f0bcc9715d6ccdad422b655c568096d37",
290
+ "deterministic": [],
291
+ "judge": {
292
+ "verdict": "pass",
293
+ "reason": "Names `cargo update -p <crate-you-bumped>` as the concrete fix (Fix section), states root cause: bumping version in Cargo.toml made Cargo.lock stale and --locked refuses (intro + final sentence). Verification shown: `cargo build --locked --all-targets`, `cargo test`. No edition/rust-version bump, no hand-editing lockfile (explicitly warns against it)."
294
+ },
295
+ "passed": true
296
+ },
297
+ {
298
+ "output": "`--locked` means \"do not touch `Cargo.lock`, just verify it already matches `Cargo.toml`.\" When you bump a version requirement in `Cargo.toml`, the lockfile still pins the old resolution, so `--locked` correctly refuses. The fix is to regenerate the lockfile once, deliberately, and commit the result — never to drop `--locked` from the CI command.\n\n## Fix it\n\n```bash\n# Option A: re-resolve just the crate you bumped (preferred — smallest diff)\ncargo update -p <crate-name>\n\n# Option B: full re-resolve if you changed several requirements\ncargo update\n```\n\nThen confirm the lockfile is now in sync with the manifest without mutating it:\n\n```bash\ncargo build --locked\n```\n\nand check the diff before committing:\n\n```bash\ngit diff Cargo.lock\n```\n\nYou should see only the changed crate (and whatever its new version transitively pulls in) bumped. Commit both `Cargo.toml` and `Cargo.lock`.\n\n## Why not just remove `--locked`\n\n`--locked` is exactly what makes CI reproducible: it fails loudly when the manifest and lockfile disagree instead of silently resolving fresh versions on the CI machine. Keep it. Regenerating `Cargo.lock` locally and committing it is the intended workflow — it's the same lockfile CI will then be verifying against.\n\n## If `cargo update` itself fails to resolve\n\nThat means the new version requirement is genuinely unsatisfiable against something else in the graph, not just a stale lockfile. Find what's holding the old version:\n\n```bash\ncargo tree -i <crate-name> # who depends on the conflicting version\ncargo update -p <crate-name> --precise <version> # test a specific target\n```\n\nRead the resolver's error — it names the conflicting requirement(s). Fix the actual constraint (e.g. a transitive dependency pinning the old major), rather than bumping the manifest version by hand to make the error go away.\n\n## Verify before reporting done\n\n```bash\ncargo build --locked\ncargo clippy --all-targets --locked -- -D warnings\ncargo test --locked\n```\n\nAll exit 0, and `git diff Cargo.lock` shows only the intended crate(s) changed. In your report, state the root cause in one sentence (e.g. \"`Cargo.lock` still pinned the pre-bump version, so `--locked` correctly rejected the out-of-date lockfile; regenerated it with `cargo update -p <crate>`\"), not just \"CI passes now.\"",
299
+ "outputSha256": "8f250f3b7cf14c8d9ff1a31befc81fe6bfde3b3967faa7c79b9e0304ec0571b8",
300
+ "promptSha256": "3f359775f43e4339282d923ee9ed7a6f0bcc9715d6ccdad422b655c568096d37",
301
+ "deterministic": [],
302
+ "judge": {
303
+ "verdict": "pass",
304
+ "reason": "Names `cargo update -p <crate-name>` and `cargo update` as the fix (Option A/B). States root cause: lockfile still pins old resolution so --locked refuses. Verification: cargo build --locked, cargo test --locked. For genuine conflict it uses cargo tree -i <crate-name> before hand-bumping, and warns against bumping manifest by hand. No edition/rust-version bump, no hand-editing Cargo.lock."
305
+ },
306
+ "passed": true
307
+ }
308
+ ]
309
+ },
310
+ {
311
+ "id": "no-clippy-allow-suppression",
312
+ "kind": "behavior",
313
+ "prompt": "cargo clippy (with the nursery group enabled) is flagging a redundant clone in my Rust code. How should I fix it?",
314
+ "strictness": "high",
315
+ "trials": 10,
316
+ "passes": 9,
317
+ "passRate": 0.9,
318
+ "passAtK": 1,
319
+ "grader": "judge",
320
+ "status": "ran",
321
+ "trialRecords": [
322
+ {
323
+ "output": "Fix the underlying ownership shape the lint is pointing at, not the lint itself.\n\nConcretely:\n\n1. Read the exact clippy diagnostic. `redundant_clone` names the specific clone expression and usually shows why it's redundant — the cloned value is either never used afterward, or is used only in a way that doesn't require ownership. That tells you which of the two cases you're in.\n\n2. Determine which it is:\n - **The original is never used again after the clone** — the clone is pure dead work. Just remove the `.clone()` and pass/move the original value instead.\n - **The original is used again, but the clone isn't actually needed for ownership** — e.g. the callee only needs a borrow (`&T`), or the value is only read. Change the call site to borrow (`&value`) and drop the clone, adjusting the callee's signature to take a reference if it currently takes `T` by value unnecessarily.\n\n3. If the callee genuinely needs an owned value and the original is genuinely used again afterward, the clone may be legitimate — in that case confirm it's the right call (a real independent copy, not just to satisfy the borrow checker) and, if so, this is a case where the lint is a false positive for your shape. Only then would a narrow, commented suppression be defensible — but check first whether restructuring (e.g. returning the original instead of cloning it, or reordering so the use comes before the move) removes the need entirely.\n\nWhat not to do:\n\n- Do not add `#[allow(clippy::redundant_clone)]` (or worse, a blanket `#[allow(clippy::all)]`) just to get a clean run. That's the exact thing this skill forbids — it silences the finding without addressing what it caught.\n- Do not reflexively replace the clone with another clone, or restructure a bunch of unrelated call sites, to make the message move elsewhere.\n\nThen verify with the full gate:\n\n```\ncargo build --all-targets\ncargo clippy --all-targets -- -D warnings\ncargo fmt --check\ncargo test\n```\n\nAll must exit 0, and your report should state the root cause in one sentence — e.g. \"the value was passed by value to a callee that only reads it, so the clone was unnecessary; changed the callee to take `&T`.\"",
324
+ "outputSha256": "54dbee89f05ea2ce46236df4247ff0453ae50dfe1a502817af84b74b77a72496",
325
+ "promptSha256": "bccf8383f1f6b9ffcd13dc810bd2d3510aa448e56e6c55eb08fd93d73d4b5711",
326
+ "deterministic": [],
327
+ "judge": {
328
+ "verdict": "pass",
329
+ "reason": "Answer identifies clone as unnecessary and to be removed (step 2: 'Just remove the .clone() and pass/move the original value instead'), shows concrete fix direction ('Change the call site to borrow (&value) and drop the clone, adjusting the callee's signature to take a reference'), and explicitly rejects #[allow(clippy::redundant_clone)]/#[allow(clippy::all)] as the fix. No fail criteria hold."
330
+ },
331
+ "passed": true
332
+ },
333
+ {
334
+ "output": "Fix the ownership shape the lint is pointing at, not the lint itself. `redundant_clone` fires when you clone a value that is never used afterward (or is about to be dropped anyway), so the clone is pure overhead.\n\nWorkflow:\n\n1. Reproduce and read the exact diagnostic:\n ```bash\n cargo clippy --all-targets -- -D warnings\n ```\n Clippy will name the exact binding and often the exact line where the unnecessary clone happens.\n\n2. Look at what the clone was doing. Usually one of these is true:\n - The original value is discarded right after, so you can just move it instead of cloning: replace `let x = y.clone(); drop(y);` (or the equivalent through a call boundary) with a move.\n - The function taking the argument should take ownership (`T`) or a reference (`&T`) rather than forcing a clone at the call site — change the signature and let the caller pass what it already has.\n - The clone was a reflexive workaround for a borrow-checker complaint; look at the compiler's own suggestion and restructure the borrow (split it, shorten a lifetime, reorder the operations) rather than continuing to clone.\n\n3. If the clone genuinely is needed — e.g. the value really must be duplicated because both the original and copy are used — then the lint is a false positive here, and the correct response is a narrowly scoped `#[allow(clippy::redundant_clone)]` **with a one-line comment explaining why the duplication is required**. That is the only acceptable suppression, and it must name the specific lint, never `#[allow(clippy::all)]`.\n\n4. Verify with the full gate:\n ```bash\n cargo build --all-targets\n cargo clippy --all-targets -- -D warnings\n cargo fmt --check\n cargo test\n ```\n All must exit 0.\n\nNote: `redundant_clone` lives in clippy's `nursery` group, which is allow-by-default, so this only surfaced because you enabled that group. Enabling a stricter group is a signal you want the finding addressed, not suppressed — so treat \"I'll just allow it\" as the fallback for a proven false positive, not the default.",
335
+ "outputSha256": "baba1985cd2984fd115a2fd02e9b18aecbf972a299be27251892a19935c23759",
336
+ "promptSha256": "bccf8383f1f6b9ffcd13dc810bd2d3510aa448e56e6c55eb08fd93d73d4b5711",
337
+ "deterministic": [],
338
+ "judge": {
339
+ "verdict": "pass",
340
+ "reason": "Answer identifies clone as unnecessary overhead (step 2) and shows concrete fixes: move instead of clone, change signature to take T or &T so caller passes what it has, restructure borrow. Suppression only for proven false positives, narrowly scoped to redundant_clone, never clippy::all. No fail criteria hold."
341
+ },
342
+ "passed": true
343
+ },
344
+ {
345
+ "output": "Read the finding first, then fix the clone — not the lint. The smallest-root-cause fix depends on which of these is actually happening:\n\n**1. The clone truly is unnecessary (most common `redundant_clone` case).**\n`nursery`'s `redundant_clone` fires when you clone a value and then never use the original again, or when the clone's lifetime is trivially short. Fix it by deleting the clone and letting the value move:\n\n```rust\n// flagged\nlet s = String::from(\"hi\");\nlet t = s.clone();\ndrop(s); // original used after the clone -> actually not redundant\n```\nvs. the genuinely redundant shape:\n```rust\n// flagged as redundant\nfn f(v: Vec<u8>) -> Vec<u8> {\n let w = v.clone();\n w // v is never touched again -> move instead\n}\n```\nFix:\n```rust\nfn f(v: Vec<u8>) -> Vec<u8> {\n v\n}\n```\n\n**2. The clone is needed for ownership, but a borrow would do.**\n`redundant_clone` sometimes surfaces because you cloned to satisfy a `&T` parameter that you could just pass a reference to. Change the local to a borrow:\n```rust\n// flagged\nprocess(data.clone());\n// fix\nprocess(&data);\n```\n\n**3. The clone is genuinely necessary.**\nIf the value must outlive the original (e.g. it's stored in a struct while the original is still borrowed, or it's an `Rc`/`Arc` shared across threads), then the clone is not redundant and clippy is pointing at something adjacent — often a surrounding move that could remove it. Look at *why* the borrow checker forced the clone in the first place; restructuring the ownership flow (splitting borrows, reordering, changing a signature from `T` to `&T`) usually removes the clone without changing behavior.\n\n**What not to do.** Do not add `#[allow(clippy::redundant_clone)]`. That silences the diagnostic without resolving whether the clone is needed — and it's the same mistake as suppressing any other clippy finding. Since `nursery` is allow-by-default, the fact that it fired here means you (or the project) opted the group in; honoring that opt-in means fixing the code, not muting the group.\n\n**Verify:** `cargo clippy --all-targets -- -D warnings` (with `nursery` enabled), `cargo build --all-targets`, `cargo fmt --check`, and `cargo test` all exit 0, and the diff touches only what the clone fix requires.",
346
+ "outputSha256": "4ab4dc10bbf6c3cb32e6e5f201d6795f477c6539561b69b41dceb6cdcd29ba56",
347
+ "promptSha256": "bccf8383f1f6b9ffcd13dc810bd2d3510aa448e56e6c55eb08fd93d73d4b5711",
348
+ "deterministic": [],
349
+ "judge": {
350
+ "verdict": "pass",
351
+ "reason": "Answer identifies the clone as unnecessary and to be removed/moved (case 1: 'delete the clone and let the value move', shows fn f(v)->v). Pass 2: concrete fix shown, e.g. 'process(data.clone())' -> 'process(&data)' and changing signature T to &T. Pass 3: explicitly says 'Do not add #[allow(clippy::redundant_clone)]'. No fail criteria hold."
352
+ },
353
+ "passed": true
354
+ },
355
+ {
356
+ "output": "Fix the actual clone, not the lint. Read clippy's full message first — it names the exact binding and, when it can, shows the two locations the value was cloned from and moved into. Then:\n\n1. Locate the clone the diagnostic points at.\n2. Determine whether the clone is genuinely needed: is the value moved later and also still borrowed, or is it duplicated pointlessly because a borrow would do?\n3. If a borrow suffices, replace the clone with a reference (`&value`) or restructure ownership so the value doesn't need to be duplicated at all. If two owners really are needed, `Rc`/`Arc` or moving the value into one owner is the real fix.\n4. Re-run clippy with the nursery group enabled to confirm the finding is gone for a reason, not suppressed.\n\nWhat not to do: do not add `#[allow(clippy::redundant_clone)]` (or a blanket `#[allow(clippy::all)]`) to silence it. `redundant_clone` is allow-by-default and only surfaces because you enabled the nursery group — once it fires, suppressing it is the same mistake as suppressing any other clippy finding, because it hides the ownership question clippy is raising rather than answering it. Likewise, do not reflexively add a *different* clone somewhere else to make the borrow checker happy; read clippy's suggested fix and the ownership shape before changing anything.\n\nIf, after looking, the clone turns out to be genuinely necessary (e.g. the value must outlive the borrow), then the lint is arguably a false positive — but the correct resolution in that case is still to understand why, not to blanket-allow the whole crate. Say so in your report and keep the suppression as narrow as possible (a targeted `#[allow(clippy::redundant_clone)]` on that one item with a one-line reason), if a suppression is truly warranted at all.\n\nAfter the change, verify with:\n\n```\ncargo build --all-targets\ncargo clippy --all-targets -- -D warnings\ncargo fmt --check\ncargo test\n```\n\nAll must exit 0. In your report, state the root cause in one sentence (e.g. \"the `name` field was cloned because it was both borrowed for logging and moved into the struct — logging can take a `&str`, so the clone was removed\"), not just \"clippy passes now.\"",
357
+ "outputSha256": "712585d2b6b48e3517c85212a52d9a0ae7aea4f886ba9902a9fa2d9f6389cf85",
358
+ "promptSha256": "bccf8383f1f6b9ffcd13dc810bd2d3510aa448e56e6c55eb08fd93d73d4b5711",
359
+ "deterministic": [],
360
+ "judge": {
361
+ "verdict": "pass",
362
+ "reason": "Step 3 concretely shows the fix: replace the clone with a reference (`&value`) or restructure ownership (Rc/Arc/move). It explicitly warns against #[allow(clippy::redundant_clone)]/#[allow(clippy::all)] as the fix. The narrow-suppression caveat is only for a genuine false positive, not the flagged finding, so it doesn't violate the fail criteria."
363
+ },
364
+ "passed": true
365
+ },
366
+ {
367
+ "output": "Read what clippy is actually pointing at first, then decide how to fix the ownership, not just the lint.\n\n`redundant_clone` (clippy's `nursery` group) fires when the compiler can see the cloned value is never used afterward — cloning it just wastes an allocation. The right fix is to remove the clone and let the original value move to where the clone was going, or restructure the borrow so no duplication is needed.\n\nConcretely:\n\n1. **Locate the flagged expression.** Run `cargo clippy --all-targets` and read the exact span — it points at the `.clone()` call and usually explains which use is redundant.\n\n2. **Try deleting the clone.** The common case is something like:\n\n ```rust\n let s = String::from(\"hi\");\n let t = s.clone(); // s is never used again\n consume(t);\n ```\n\n Since `s` is never used after, move it instead:\n\n ```rust\n let s = String::from(\"hi\");\n consume(s);\n ```\n\n If the compiler then errors with a \"use of moved value\" or a borrow conflict, that tells you the original value *is* needed later, which usually means the clone was not actually redundant in that path — double-check the lint span against your actual control flow rather than trusting a stale read.\n\n3. **If a borrow is what forces the clone**, fix the borrow shape instead of cloning past it. Typical fixes:\n - Take `&str`/`&T` in the callee where it previously took an owned `String`/`T`, so you can pass a reference.\n - Reorder use of the value so the borrow ends before the move.\n - Split a struct field borrow so you are not cloning an entire owned field just to pass one piece of it.\n\n4. **When a clone genuinely is needed**, leave it — but then the lint shouldn't be firing; if it still does, the clone is provably dead and step 2 applies.\n\nDo **not** add `#[allow(clippy::redundant_clone)]` (or a blanket `#[allow(clippy::all)]`) to silence it. That just hides the wasted allocation without fixing the ownership the checker is correctly pointing out. And don't reflexively add a *different* clone somewhere else to work around a borrow error the fix introduces — resolve the ownership properly.\n\nAfter the change, confirm:\n\n```bash\ncargo build --all-targets\ncargo clippy --all-targets -- -D warnings\ncargo fmt --check\ncargo test\n```\n\nAll should exit 0. If removing the clone broke a call site, the fix is to adjust that call site's ownership/borrow, not to restore the clone.\n\nRoot cause to state in your report: e.g. \"clippy flagged a clone of a value never used again; replaced `.clone()` with a move (and changed the callee to take `&T` where it only read the value), so the allocation is gone.\" Not just \"silenced the lint.\"",
368
+ "outputSha256": "4ceedc96d5bab5ee72661066f2dd08913d8b9104a7c76b3f26f845f3ecadd524",
369
+ "promptSha256": "bccf8383f1f6b9ffcd13dc810bd2d3510aa448e56e6c55eb08fd93d73d4b5711",
370
+ "deterministic": [],
371
+ "judge": {
372
+ "verdict": "pass",
373
+ "reason": "Answer says remove the clone and move/borrow instead (step 2 shows deleting `.clone()` and moving `s`; step 3 shows taking `&str`/`&T` in the callee to pass a reference). It explicitly rejects `#[allow(clippy::redundant_clone)]`/`#[allow(clippy::all)]` as the fix. No fail criteria hold."
374
+ },
375
+ "passed": true
376
+ },
377
+ {
378
+ "output": "The short answer: don't reach for `#[allow(clippy::redundant_clone)]`. Read what the lint is pointing at and fix the ownership, because that's what \"redundant clone\" means — the clone isn't needed to satisfy the borrow checker, and removing it is the fix.\n\nA quick note on why you're only seeing this now: `redundant_clone` lives in clippy's `nursery` group, which is allow-by-default. It only fires because you explicitly enabled nursery, so this is the same class of finding as any other clippy lint — fix the underlying issue, don't suppress it.\n\nWhat the lint is actually telling you:\n\nThe clone produces a value that is never mutated or needed as a separate owned copy on any path through the code. In practice this shows up in a few shapes:\n\n- The clone result is used once and immutably, or dropped before any divergence — a `&` borrow of the original would satisfy the same call.\n- The clone is followed by code that immediately overwrites the copy, so the duplication was pointless.\n- You cloned to \"get around\" a borrow conflict, but the borrow checker would have been fine with the original binding if borrows were scoped/ordered better.\n\nHow to fix, concretely:\n\n1. Delete the `.clone()` and let the compiler tell you what actually breaks. Often nothing does — the borrow of the original is sufficient, and the borrow checker accepts it.\n2. If a borrow conflict does appear, fix the borrow structure rather than reintroducing the clone: split the borrow, shorten a borrow's scope so it ends before the conflicting use, restructure so you take a `&`/`&mut` where the callee only needs a reference, or reorder the operations. `rustc`'s suggestion on the resulting `E0502`/`E0499` usually names the exact conflicting borrows — follow it.\n3. Only keep a clone (or add one) when the value genuinely must be an independent owned copy — e.g. it's moved into a spawned task, a different-ownership container, or a return value that outlives the borrow. In that case the clone is legitimate and the lint generally won't fire; if it does and you've confirmed the ownership need, a narrowly scoped `#[allow(clippy::redundant_clone)]` with a comment explaining why the copy is required is defensible, but that's the last resort, not the first move.\n\nWhat not to do:\n\n- Don't add `#[allow(clippy::redundant_clone)]` (or worse, `#[allow(clippy::all)]`) just to make the checker quiet.\n- Don't swap in an `.unwrap()` or a different clone nearby to dodge the borrow error; you'd be trading one smell for another.\n- Don't bump the edition or a dependency version to make it go away — the lint isn't an edition/MSRV problem.\n\nAfter the change, verify with the full gate and confirm nothing else regressed:\n\n```\ncargo build --all-targets\ncargo clippy --all-targets -- -D warnings\ncargo fmt --check\ncargo test\n```\n\nAll four must exit 0. If removing the clone makes `cargo build` pass but changes behavior (a shared value that was supposed to be independent), that's a signal the clone was semantically load-bearing after all — go back and confirm the ownership intent before deciding it was truly redundant.\n\nOne-sentence summary for your report: \"Removed a redundant `.clone()` that copied a value never used as an independent owner; the borrow of the original satisfies the call, and build/clippy/fmt/test all pass.\" State the root cause, not just \"build now passes.\"",
379
+ "outputSha256": "7a4558145c1233c94b3866913b1c13269955ca12171a9898a8d071aed3503b3b",
380
+ "promptSha256": "bccf8383f1f6b9ffcd13dc810bd2d3510aa448e56e6c55eb08fd93d73d4b5711",
381
+ "deterministic": [],
382
+ "judge": {
383
+ "verdict": "pass",
384
+ "reason": "Answer says don't use #[allow(clippy::redundant_clone)] and fix ownership; concrete fix: 'Delete the .clone()' and 'take a &/&mut where the callee only needs a reference'. It warns against #[allow(clippy::all)]. The narrow allow is only a last resort for genuinely needed clones, not the flagged redundant one, so fail criteria don't hold."
385
+ },
386
+ "passed": true
387
+ },
388
+ {
389
+ "output": "Fix the clone, not the lint. A `clippy::redundant_clone` finding means the compiler can see that the cloned value is never used after the clone point — the clone is pure overhead, not a necessary duplication.\n\n**What to do:**\n\n1. Read the exact span clippy points at. It names the binding that was cloned and shows where the original goes unused afterward.\n2. Remove the `.clone()` and see if the code still borrows correctly. Usually the original value is already owned at that point or can be borrowed (`&value`) instead.\n3. If removing it *does* trigger a borrow-checker error (`E0502`/`E0499`), that's a real signal that the surrounding ownership shape needs restructuring — shorten a borrow, reorder statements, or take a reference earlier. Don't reinstate the clone to paper over it.\n4. If the original genuinely escaped scope and you truly need two owning copies, then the clone isn't redundant — check whether clippy's suggestion is actually wrong for your case, but reach that conclusion by reasoning about the ownership, not by suppressing the lint.\n\n**What not to do:**\n\n- Do **not** add `#[allow(clippy::redundant_clone)]`, and never a blanket `#[allow(clippy::all)]`. That hides the finding without addressing what it found.\n- Do **not** swap `.clone()` for something equivalent like `to_owned()` just to dodge the lint name — if the value is genuinely redundant, the fix is to drop the duplication, not disguise it.\n\n**Verify:** re-run `cargo clippy --all-targets -- -D warnings` (with your nursery group still enabled), plus `cargo build --all-targets`, `cargo fmt --check`, and `cargo test` — all four must exit 0. Then state the root cause in one sentence (e.g., \"removed a clone of `config` that was dead after the move point\"), not just \"lint passes.\"\n\nNote: `redundant_clone` lives in clippy's `nursery` group, which is allow-by-default — it only fires because you've enabled nursery. The fix is identical regardless of which group surfaced it: address the actual redundancy.",
390
+ "outputSha256": "d110b9a1cb7b5e121058cf0c462b4cd92bac433c4642d41aaa180627f35bd64a",
391
+ "promptSha256": "bccf8383f1f6b9ffcd13dc810bd2d3510aa448e56e6c55eb08fd93d73d4b5711",
392
+ "deterministic": [],
393
+ "judge": {
394
+ "verdict": "pass",
395
+ "reason": "Answer says 'Fix the clone, not the lint' and step 2 shows concrete fix: 'Remove the .clone()' and 'can be borrowed (&value) instead'. Step 3 restructures ownership (shorten borrow, reorder). Explicitly rejects #[allow(clippy::redundant_clone)] and #[allow(clippy::all)]. No fail criteria met."
396
+ },
397
+ "passed": true
398
+ },
399
+ {
400
+ "output": "Don't silence it — fix the underlying ownership shape. A `redundant_clone` finding means clippy can prove the cloned value is never used afterward on that path, so the clone is pure waste. Reaching for `#[allow(clippy::redundant_clone)]` (or a blanket `#[allow(clippy::all)]`) just hides what the lint correctly found and is exactly the suppression this skill forbids.\n\nWork through it like this:\n\n1. **Read the exact span.** `cargo clippy` points at the specific `.clone()` call. Look at how the original value (the one being cloned) is used after that line, and how the clone is used. The lint fires when the binding being cloned is dead afterward — i.e. the clone could be replaced by a move.\n\n2. **Try the move first.** Most of the time the fix is to drop `.clone()` and let ownership transfer:\n ```rust\n // clippy: redundant_clone\n let s = String::from(\"hi\");\n let t = s.clone();\n consume(t);\n // s never used again\n\n // fix: just move it\n let s = String::from(\"hi\");\n consume(s);\n ```\n That's the smallest correct change and it's what the lint is inviting.\n\n3. **If the move doesn't compile, understand why** before doing anything else. The remaining uses of the original value are the real constraint. The compiler will tell you which borrow conflicts. Then choose the right restructure rather than reflexively cloning:\n - Split the borrow / reorder statements so the original's last use precedes the move.\n - Take a reference instead of an owned value, if the callee only reads it (`&s` rather than cloning to pass ownership).\n - If the value genuinely must be duplicated per `rules/patterns.mdc` (independent owners on divergent paths), a clone is legitimate — but confirm that's the case rather than assuming it.\n\n4. **Re-verify**, since clippy's `redundant_clone` is in the `nursery` group (allow-by-default, so it only fires because you enabled nursery):\n ```\n cargo build --all-targets\n cargo clippy --all-targets -- -D warnings\n cargo fmt --check\n cargo test\n ```\n All four must exit 0.\n\nThe red-flag version of this is \"I'll add `#[allow(clippy::redundant_clone)]` so the linter stops complaining\" — that closes the diagnostic without asking whether the clone was ever needed. The lint already answered that question for you: it was not. Remove it (or replace it with a move/reference), and if a clone really is required, that fact will fall out of the ownership analysis in step 3 rather than being asserted to make the warning go away.",
401
+ "outputSha256": "c756347bf0e496be1d89048dc770ca0dde8f0ec10c54e8dd6f32845965ddce05",
402
+ "promptSha256": "bccf8383f1f6b9ffcd13dc810bd2d3510aa448e56e6c55eb08fd93d73d4b5711",
403
+ "deterministic": [],
404
+ "judge": {
405
+ "verdict": "pass",
406
+ "reason": "Answer says clone is 'pure waste' and should be removed (criterion 1). Concrete fix shown: drop .clone() and move, e.g. `consume(s)` instead of `consume(t)`, plus 'Take a reference instead of an owned value ... `&s`' (criterion 2). Explicitly rejects `#[allow(clippy::redundant_clone)]`/`#[allow(clippy::all)]` as suppression (criterion 3). No fail criteria hold."
407
+ },
408
+ "passed": true
409
+ },
410
+ {
411
+ "output": "Read the diagnostic the compiler gave you before changing anything: `clippy::redundant_clone` names the exact expression it believes already owns a value, so the fix is almost always to delete the clone or restructure ownership so it isn't needed — not to suppress the lint. Walk through the snippet clippy points at and ask where that value actually needs to be owned versus borrowed. Common real fixes: move the value into the call instead of cloning it, take a reference (`&T`) where the callee only reads, or reorder the borrows so the original binding is dropped before the consumer needs it. Only add `.clone()` deliberately when the value genuinely needs to be duplicated (e.g. both the original and the consumer keep it) — and if that's the case, don't suppress the lint blindly; it usually means you can make the ownership intent explicit instead.\n\nConcretely, the shape clippy is flagging is usually one of these:\n\n```rust\n// flagged: v is not used after this point, so the clone is pure overhead\nlet s = format!(\"{v:?}\");\nconsume(v.clone());\n\n// fix: drop the clone and move v\nlet s = format!(\"{v:?}\");\nconsume(v);\n```\n\nor, when the callee only needs to read:\n\n```rust\n// flagged: callee takes &T but we cloned first\ntake_ref(owned.clone());\n\n// fix: hand it a borrow, no clone\ntake_ref(&owned);\n```\n\nAfter the change, re-run:\n\n```\ncargo clippy --all-targets -- -D warnings\ncargo build --all-targets\ncargo test\n```\n\nAll three must exit 0. Then report the root cause in one sentence — e.g. \"Removed a redundant clone in `foo`: `v` was moved into `consume` immediately after, so the duplicate was never observed\" — not just \"clippy passes now.\"\n\nTwo things to avoid: never add `#[allow(clippy::redundant_clone)]` (or a blanket `#[allow(clippy::all)]`) to make it go away, and don't replace the clone with a reflexive `.unwrap()` or an unrelated restructuring just to silence it. Note also that `redundant_clone` lives in clippy's `nursery` group, so it only fires because you enabled that group — that's not a reason to treat the finding as noise; it's pointing at real duplicated work.\n\nIf, after reading the diagnostic, you conclude the clone really is required (both the original and the consumer keep the value), that's the rare case where the lint is wrong for your code — but even then, prefer making that ownership explicit (e.g. a comment or a clearly named binding) over a bare suppression, and say so in your report rather than silently allowing it.",
412
+ "outputSha256": "b709bb66b79935a2b12e44635c5e5618ebe391a49115537fab66fcc287b674e6",
413
+ "promptSha256": "bccf8383f1f6b9ffcd13dc810bd2d3510aa448e56e6c55eb08fd93d73d4b5711",
414
+ "deterministic": [],
415
+ "judge": {
416
+ "verdict": "pass",
417
+ "reason": "Answer says delete the clone or restructure ownership, not suppress. Concrete fixes shown: `consume(v)` instead of `consume(v.clone())`, and `take_ref(&owned)` instead of `take_ref(owned.clone())`. Explicitly warns never to add `#[allow(clippy::redundant_clone)]` or `#[allow(clippy::all)]`. All pass criteria met; no fail criteria."
418
+ },
419
+ "passed": true
420
+ },
421
+ {
422
+ "output": "Fix the underlying ownership shape rather than suppressing the lint. `redundant_clone` firing means the compiler can prove the cloned value is never used afterward — so the clone is genuinely dead weight, not a judgment call about whether duplication is needed.\n\nConcretely, work through it in this order:\n\n**1. Read the exact site clippy names.** The diagnostic points at the binding that was cloned. Ask: after this clone, is the original binding used again on any path? If not (which is what the lint has already proven), the clone can usually just be deleted, and the consuming call can take ownership of the original directly:\n\n```rust\n// Before -- clone only exists because `s` was borrowed later\nlet s = String::from(\"x\");\nlet t = s.clone();\ndo_something(t);\n// ...s never used again\n\n// After\nlet s = String::from(\"x\");\ndo_something(s);\n```\n\n**2. If the original *is* used later, the fix is a borrow, not a clone.** Pass `&s` instead of cloning, provided the callee's signature accepts a reference. If the callee demands an owned `String`, that's a signal the API is over-constrained — consider whether it should take `impl AsRef<str>` / `&str` instead. Only change the signature if it's your type; don't churn a third-party API.\n\n**3. If neither works, keep the clone and explain why in the report** — e.g. the value must outlive the borrow, or two independent owned copies are genuinely required by the logic. In that narrow case, a *scoped, justified* allow with a comment is defensible:\n\n```rust\n#[allow(clippy::redundant_clone)] // both copies are consumed by separate tasks\nlet t = s.clone();\n```\n\nBut this is the exception, not the first move.\n\n**What not to do:**\n- Do not add a blanket `#[allow(clippy::all)]` or a crate-level suppression to make the finding vanish — that's the one thing the skill's rules forbid outright.\n- Do not reflexively replace the clone with `.unwrap()` or some other silencer; that trades one masked issue for another.\n- Do not disable the `nursery` group to dodge the lint.\n\n**Then verify:** rerun `cargo clippy --all-targets -- -D warnings`, plus `cargo build --all-targets`, `cargo fmt --check`, and `cargo test`, and confirm all exit 0. Since removing a clone can change borrow lifetimes, the build step after the edit is what confirms the restructure actually holds.\n\nThe short version: `redundant_clone` is one of the rare clippy lints where the fix is almost always \"delete the clone\" — treat a suppression as a last resort requiring a stated reason, not a shortcut.",
423
+ "outputSha256": "d327aecc6312c6a64c7450457817b64fa69915950e8d698d0819b0f510e5533d",
424
+ "promptSha256": "bccf8383f1f6b9ffcd13dc810bd2d3510aa448e56e6c55eb08fd93d73d4b5711",
425
+ "deterministic": [],
426
+ "judge": {
427
+ "verdict": "fail",
428
+ "reason": "Pass criteria 1 and 2 are met (delete clone / pass &s instead of cloning, shown concretely). But criterion 3 fails: step 3 explicitly endorses `#[allow(clippy::redundant_clone)]` with a comment as 'defensible', which is exactly fail criterion 1. The answer relies on an allow as a fallback fix."
429
+ },
430
+ "passed": false
431
+ }
432
+ ]
433
+ }
434
+ ],
435
+ "verdict": "fail",
436
+ "scope": "bundled",
437
+ "skillDigest": "9057abae2b9081e6d4b7b2f0855cfc6d36105608e12a7d3c845fe960faae5a4f",
438
+ "catalogDigest": "58f55f0866f2076d0959c6ddb3326a6ed904e5fb06de729cf465181d86b85e90",
439
+ "judgePromptVersion": "2026-09-25.1",
440
+ "runner": "deepseek",
441
+ "model": "deepseek-chat",
442
+ "runnerPromptVersion": "2026-09-25.1",
443
+ "recordedAt": "2026-09-25T22:09:50.207Z",
444
+ "judge": "deepseek",
445
+ "judgeModel": "deepseek-chat"
446
+ },
447
+ {
448
+ "schemaVersion": "1.0.0",
449
+ "skillId": "rust/rust-code-review",
450
+ "strictness": "high",
451
+ "trials": 10,
452
+ "triggerAccuracy": {
453
+ "truePositive": 4,
454
+ "falsePositive": 1,
455
+ "positives": 6,
456
+ "negatives": 6
457
+ },
458
+ "evidence": "authored",
459
+ "scenarios": [
460
+ {
461
+ "id": "trigger-positive-1",
462
+ "kind": "trigger-positive",
463
+ "prompt": "Review this Rust pull request for unwrap panics on untrusted input",
464
+ "strictness": "high",
465
+ "trials": 1,
466
+ "passes": 1,
467
+ "passRate": 1,
468
+ "passAtK": 1,
469
+ "grader": "trigger-rank-fork-family",
470
+ "status": "ran",
471
+ "deterministic": true
472
+ },
473
+ {
474
+ "id": "trigger-positive-2",
475
+ "kind": "trigger-positive",
476
+ "prompt": "Check this Rust diff for unsafe blocks missing a Safety comment",
477
+ "strictness": "high",
478
+ "trials": 1,
479
+ "passes": 1,
480
+ "passRate": 1,
481
+ "passAtK": 1,
482
+ "grader": "trigger-rank-fork-family",
483
+ "status": "ran",
484
+ "deterministic": true
485
+ },
486
+ {
487
+ "id": "trigger-positive-3",
488
+ "kind": "trigger-positive",
489
+ "prompt": "Does this Rust change block the tokio runtime inside an async fn?",
490
+ "strictness": "high",
491
+ "trials": 1,
492
+ "passes": 1,
493
+ "passRate": 1,
494
+ "passAtK": 1,
495
+ "grader": "trigger-rank-fork-family",
496
+ "status": "ran",
497
+ "deterministic": true
498
+ },
499
+ {
500
+ "id": "trigger-positive-4",
501
+ "kind": "trigger-positive",
502
+ "prompt": "Review this Rust concurrency change for unjoined spawned tasks",
503
+ "strictness": "high",
504
+ "trials": 1,
505
+ "passes": 0,
506
+ "passRate": 0,
507
+ "passAtK": 0,
508
+ "grader": "trigger-rank-fork-family",
509
+ "status": "ran",
510
+ "deterministic": true
511
+ },
512
+ {
513
+ "id": "trigger-positive-5",
514
+ "kind": "trigger-positive",
515
+ "prompt": "Check for unnecessary clones used as a borrow-checker workaround in this Rust diff",
516
+ "strictness": "high",
517
+ "trials": 1,
518
+ "passes": 1,
519
+ "passRate": 1,
520
+ "passAtK": 1,
521
+ "grader": "trigger-rank-fork-family",
522
+ "status": "ran",
523
+ "deterministic": true
524
+ },
525
+ {
526
+ "id": "trigger-positive-6",
527
+ "kind": "trigger-positive",
528
+ "prompt": "This error enum has a variant that stores just a String instead of wrapping the original error -- is that going to bite us later, can you check the diff?",
529
+ "strictness": "high",
530
+ "trials": 1,
531
+ "passes": 0,
532
+ "passRate": 0,
533
+ "passAtK": 0,
534
+ "grader": "trigger-rank-fork-family",
535
+ "status": "ran",
536
+ "deterministic": true
537
+ },
538
+ {
539
+ "id": "trigger-negative-1",
540
+ "kind": "trigger-negative",
541
+ "prompt": "Review this Rust code and also fix the bugs you find",
542
+ "strictness": "high",
543
+ "trials": 1,
544
+ "passes": 1,
545
+ "passRate": 1,
546
+ "passAtK": 1,
547
+ "grader": "trigger-rank-fork-family",
548
+ "status": "ran",
549
+ "deterministic": true
550
+ },
551
+ {
552
+ "id": "trigger-negative-2",
553
+ "kind": "trigger-negative",
554
+ "prompt": "Review this Python code for SQL injection",
555
+ "strictness": "high",
556
+ "trials": 1,
557
+ "passes": 1,
558
+ "passRate": 1,
559
+ "passAtK": 1,
560
+ "grader": "trigger-rank-fork-family",
561
+ "status": "ran",
562
+ "deterministic": true
563
+ },
564
+ {
565
+ "id": "trigger-negative-3",
566
+ "kind": "trigger-negative",
567
+ "prompt": "Review this Go diff for goroutine leaks",
568
+ "strictness": "high",
569
+ "trials": 1,
570
+ "passes": 1,
571
+ "passRate": 1,
572
+ "passAtK": 1,
573
+ "grader": "trigger-rank-fork-family",
574
+ "status": "ran",
575
+ "deterministic": true
576
+ },
577
+ {
578
+ "id": "trigger-negative-4",
579
+ "kind": "trigger-negative",
580
+ "prompt": "Implement a bounded async worker pool in this Rust crate using tokio",
581
+ "strictness": "high",
582
+ "trials": 1,
583
+ "passes": 1,
584
+ "passRate": 1,
585
+ "passAtK": 1,
586
+ "grader": "trigger-rank-fork-family",
587
+ "status": "ran",
588
+ "deterministic": true
589
+ },
590
+ {
591
+ "id": "trigger-negative-5",
592
+ "kind": "trigger-negative",
593
+ "prompt": "Run a general security review on this codebase",
594
+ "strictness": "high",
595
+ "trials": 1,
596
+ "passes": 1,
597
+ "passRate": 1,
598
+ "passAtK": 1,
599
+ "grader": "trigger-rank-fork-family",
600
+ "status": "ran",
601
+ "deterministic": true
602
+ },
603
+ {
604
+ "id": "trigger-negative-6",
605
+ "kind": "trigger-negative",
606
+ "prompt": "Review this Rust diff for naming conventions and formatting only",
607
+ "strictness": "high",
608
+ "trials": 1,
609
+ "passes": 0,
610
+ "passRate": 0,
611
+ "passAtK": 0,
612
+ "grader": "trigger-rank-fork-family",
613
+ "status": "ran",
614
+ "deterministic": true
615
+ },
616
+ {
617
+ "id": "read-only-unwrap-review",
618
+ "kind": "behavior",
619
+ "prompt": "Review this Rust diff: a new function calls `response.json::<Payload>().await.unwrap()` on an HTTP client's response. What do you find?",
620
+ "strictness": "high",
621
+ "trials": 10,
622
+ "passes": 10,
623
+ "passRate": 1,
624
+ "passAtK": 1,
625
+ "grader": "judge",
626
+ "status": "ran",
627
+ "trialRecords": [
628
+ {
629
+ "output": "**Finding — `.unwrap()` on a fallible network operation**\n\n`response.json::<Payload>().await.unwrap()`\n\nThis is a panic on fallible, attacker-influenced input — the exact pattern Step 2's first bullet targets.\n\n**Why it matters**\n`reqwest::Response::json::<T>()` returns `Result<T, reqwest::Error>`. That `Result` can be `Err` for reasons entirely outside this function's control:\n\n- the connection dropped mid-body or the response body was truncated;\n- the server returned a non-2xx status (`.json()` still attempts to deserialize the body, so an HTML error page or empty body fails here);\n- the payload is valid JSON but does not match `Payload`'s shape — a renamed/added/removed field, a type change (`\"3\"` vs `3`), or a schema evolution on the server side;\n- malformed JSON from a proxy, CDN, or hostile intermediary.\n\nAny of these panics the caller. In a request-handling task this is a per-request DoS vector (one bad response kills the worker), and in a `tokio::spawn`'d task it becomes a silent task death that the join site may never observe. This is not a \"the caller always validates first\" case — the failure originates in the network round-trip, not in a precondition this function can check.\n\n**Fix direction**\nPropagate it rather than panicking. If the enclosing function returns `Result<_, E>`:\n\n```rust\nlet payload = response.json::<Payload>().await?;\n```\n\nand make sure `E` can carry the underlying cause — via `?` into a `#[from] reqwest::Error` variant, or a `#[source]`-annotated variant if the error type wraps rather than converts. If this function deliberately returns a non-`Result` type, that is itself the design question to raise: a function whose body can fail on I/O should surface that in its signature.\n\n**Note on scope**\nOnly the `.unwrap()` is asserted here. Whether the `Payload` type is robust to schema drift, and whether `response` was checked for a success status before this line, are both worth reading in the surrounding code — a status check that discards the body on non-2xx would not make the `unwrap()` safe, but it would change which failure modes are realistically reachable. I have not seen that context in this diff.\n\n**Also worth confirming at this call site**\n- Is this `.await` inside an `async fn` that holds a `std::sync::Mutex` guard or other non-`Send`/blocking resource across it? That is a separate finding from the `unwrap()`, not implied by it.\n- If the enclosing task was `tokio::spawn`'d, is there a visible join or channel? A panic here turns into a dropped `JoinHandle` result that nobody reads, which makes the failure invisible rather than loud.\n\nThis review is read-only; no file was modified. The finding above is the fix direction only — apply the `?`-propagation change (and any error-variant addition it requires) yourself, in an edit step.",
630
+ "outputSha256": "6e1c8a715bf992b8e2513a4f5fb002f290a93496a96f645fa3907f2f9a68b2d8",
631
+ "promptSha256": "9ed84486979974f1dc74a79abfd6ce71bdb6ec0039934bb03cd5cbd92d7aca80",
632
+ "deterministic": [],
633
+ "judge": {
634
+ "verdict": "pass",
635
+ "reason": "Names the exact call `response.json::<Payload>().await.unwrap()` as the problem (criterion 1). Explains network response is fallible/untrusted input: truncated body, non-2xx, schema mismatch, malformed JSON from proxy/CDN — panics the caller, per-request DoS (criterion 2). Fix direction: propagate with `?` into Result, `#[from] reqwest::Error` variant, shown as finding only; explicitly states read-only, no file modified (criterion 3). No edit claimed."
636
+ },
637
+ "passed": true
638
+ },
639
+ {
640
+ "output": "**Finding — file:line (the new function): `response.json::<Payload>().await.unwrap()`**\n\n- **Pattern:** `.unwrap()` on a `Result` produced from a network response body. The `.json::<Payload>()` call is fallible for *both* reasons at once: the I/O of reading the body can fail, and the deserialization of an attacker- or peer-controlled body into `Payload` can fail. Both paths panic here.\n- **Why it matters:** This is a panic reachable from untrusted input. A malformed, truncated, empty, or unexpected-schema response body — or a dropped connection mid-body — will unwind instead of returning an error. In a request-handling path that is a DoS vector (one bad upstream response kills the task, or the process if panic=abort). It also means the caller gets no typed error to react to; a transient network failure becomes indistinguishable from a programming bug.\n- **Fix direction (do not apply):** propagate with `?` into the surrounding function's error type — `let payload: Payload = response.json().await?;`. If that error type doesn't yet carry a deserialization/transport variant, add one (with `#[source]`/`#[from]` so the underlying cause is preserved) rather than stringifying. If this function genuinely cannot return a `Result`, that is the design question to raise — a fallible network call usually means the signature should change, not that the failure should be swallowed.\n\n**Two related things to check while you're in this function (flag only if present in the diff):**\n\n- If the surrounding `async fn` holds a `std::sync::Mutex` guard or calls `std::thread::sleep`/blocking `std::fs` I/O across this `.await`, that is a separate executor-stall finding.\n- If `Payload`'s fields include numeric types parsed from the untrusted body, check any subsequent arithmetic on them for unchecked `+`/`*` rather than `checked_*`/`saturating_*` chosen for the operation's real semantics.\n\n**Not a finding:** the presence of a `.clone()` or indexing elsewhere in the function is out of scope unless the diff adds it; don't pad the report.\n\nI did not edit anything — this is report-only. One caveat: I'm reading a description of the call, not the actual file, so I've given the finding by pattern rather than an exact line number; substitute the real `file:line` when you file it.",
641
+ "outputSha256": "2c64efe1772b1920e8c6e7a4593285f67da3815fc3f8ed77d1adb95bf7a33842",
642
+ "promptSha256": "9ed84486979974f1dc74a79abfd6ce71bdb6ec0039934bb03cd5cbd92d7aca80",
643
+ "deterministic": [],
644
+ "judge": {
645
+ "verdict": "pass",
646
+ "reason": "Names the exact call `response.json::<Payload>().await.unwrap()` as the finding (criterion 1). Explains the body is untrusted/fallible input — malformed, truncated, empty, or unexpected-schema body panics instead of returning an error (criterion 2). Gives concrete fix direction: propagate with `?` (`let payload: Payload = response.json().await?;`) or add an error variant, explicitly marked 'do not apply' (criterion 3). States 'I did not edit anything — this is report-only', so no fail criterion holds."
647
+ },
648
+ "passed": true
649
+ },
650
+ {
651
+ "output": "**Finding — `src/<file>.rs:1` (the new function; the diff shows `.unwrap()` inline on the `.json::<Payload>().await` result): `.unwrap()` on a fallible, network-derived `Result`. Risk: panic.**\n\nCategory: panics on fallible/untrusted input.\n\n- `response.json::<Payload>()` returns `Result<Payload, _>` whose error covers both a transport-level failure (body could not be read) and a decode-level failure (body was not valid JSON, or did not match `Payload`). The body comes from a remote peer, so `Payload` is attacker-influenced input, not a value this code controls.\n- `.unwrap()` turns either failure into a panic. In a request-handling context that panic aborts the task and, depending on the runtime and panic strategy, can take down the worker or the process — a remote party can trigger it just by returning a non-JSON body or a schema that no longer matches `Payload`. That is a denial-of-service surface, not just an ergonomics nit.\n- This is diff-introduced, not pre-existing: the pattern appears in the new function.\n\n**Fix direction (not applied — this review is read-only):** propagate instead of panicking. Either\n- return the error with `?`, mapping it into the caller's error type (`response.json::<Payload>().await?`), or\n- if the caller genuinely must distinguish \"transport failed\" from \"body didn't parse\", match on the `reqwest::Error` and add a `#[source]`-wrapping variant to the module's error enum so the cause is preserved for the caller rather than flattened.\n\nDo not paper over it by validating \"the caller always sends valid JSON\" — that invariant is not visible at this call site and is exactly the assumption an attacker violates.\n\n**Related checks I could not settle from the one-line description — worth confirming at the real call site:**\n- If the enclosing function is `async`, no blocking call was introduced, but the `.await` here is fine; if the result is fed into indexing or arithmetic on `Payload` fields downstream, check those against the unchecked-indexing / unchecked-arithmetic rules separately.\n- If the surrounding code holds a `std::sync::Mutex` guard across this `.await`, that is a second, distinct finding (executor stall); the diff excerpt does not show it.\n\n**Pre-existing vs. introduced:** the `.unwrap()` is introduced by this diff. Anything flagged in surrounding unchanged lines in the same file should be reported separately as pre-existing.\n\nI have not edited any file. Every finding above names the line as far as the excerpt allows (the diff shows it at the top of the new function); in a full review the exact `file:line` would be taken from the diff hunk headers.",
652
+ "outputSha256": "1ab418cd00c3e29c4c6598c52952e18e4ff8206965672deb350bac8366509088",
653
+ "promptSha256": "9ed84486979974f1dc74a79abfd6ce71bdb6ec0039934bb03cd5cbd92d7aca80",
654
+ "deterministic": [],
655
+ "judge": {
656
+ "verdict": "pass",
657
+ "reason": "Names the specific call: '.unwrap() on a fallible, network-derived Result' on 'response.json::<Payload>()'. Explains untrusted/fallible input: 'body comes from a remote peer... attacker-influenced', panic 'can take down the worker or the process'. Fix direction: 'return the error with `?`' or match on reqwest::Error, explicitly 'not applied — this review is read-only'. No edit made."
658
+ },
659
+ "passed": true
660
+ },
661
+ {
662
+ "output": "**Finding — src: the new function's `.unwrap()` on `response.json::<Payload>().await`**\n\n- **Pattern:** `.unwrap()` on the `Result` returned by `reqwest`'s (or equivalent) `Response::json::<Payload>()`.\n- **Why it matters:** `json()` is fallible on at least two attacker-influenced paths: (1) the response body bytes may not deserialize into `Payload` (schema drift, truncated body, error page returned with a 200, or a hostile/malformed payload), and (2) the body read itself can fail (connection reset mid-body, timeout). Any of these panics the task/thread. Per this skill's focus list, `.unwrap()` on a `Result` built from network input is a panic reachable from untrusted input — a DoS on the caller if the panic is not caught upstream, and at minimum a crash that bypasses the error handling the rest of the function presumably has.\n- **Fix direction (do not apply — read-only):** propagate with `?` into the enclosing function's error type. If the enclosing error enum doesn't yet carry a deserialization/transport cause, add a variant that wraps the underlying error with `#[from]` or `#[source]` so the original `reqwest::Error` is not flattened to a `String` (see the error-handling check below).\n\n**Related checks on the same line, in priority order:**\n\n1. **Status code not checked.** If the diff reaches `json()` without first checking `response.status()` (e.g. `.error_for_status()?`), a 4xx/5xx body is being parsed as if it were `Payload`. That is the same class of bug as the unwrap: the failure surfaces as a confusing deserialization panic instead of a clear HTTP error. Flag it separately if present.\n2. **Error source loss.** If the fix direction above leads to a new error variant that does `err.to_string()`, that drops the source error — same finding category as the unwrap, worth calling out now so the fix does not reintroduce it.\n3. **Async context.** Confirm the call is in an `async fn` using an async client (not a blocking `reqwest::blocking` call inside `async`), and that no `std::sync` guard is held across this `.await`.\n\n**What I cannot confirm from the description alone:** whether `Payload`'s `Deserialize` impl has any panicking path (e.g. a custom `deserialize_with` that unwraps internally), and whether the caller of this function catches panics. Both are worth a glance at the diff, but the `.unwrap()` is the finding regardless.\n\n**Verdict:** one diff-introduced finding, high confidence — `.unwrap()` on a fallible network-derived `Result`. If the diff also skips a status check, that is a second finding. No source file was modified; this is report-only.",
663
+ "outputSha256": "e1003a4eea1263cefba35d04305f6a8168695a80b51e5a24f0e564154ae6f8cc",
664
+ "promptSha256": "9ed84486979974f1dc74a79abfd6ce71bdb6ec0039934bb03cd5cbd92d7aca80",
665
+ "deterministic": [],
666
+ "judge": {
667
+ "verdict": "pass",
668
+ "reason": "Names the specific call: '.unwrap() on response.json::<Payload>().await' (Pattern bullet). Explains fallibility: body may not deserialize or read can fail, 'Any of these panics the task/thread', untrusted input DoS. Fix direction: 'propagate with ? into the enclosing function's error type', presented as 'do not apply — read-only'. No edit claimed; states 'No source file was modified; this is report-only.'"
669
+ },
670
+ "passed": true
671
+ },
672
+ {
673
+ "output": "Finding — panics on fallible network input.\n\n`src/.../...rs:<line of the .unwrap()>` — `response.json::<Payload>().await.unwrap()`.\n\nWhat it is: `.unwrap()` on the `Result` returned by `json::<Payload>()`, which is fallible purely because the input is a live HTTP response. It fails whenever the body is not valid JSON for `Payload` — a truncated or empty body, a non-JSON error page returned with a 200, a schema change on the server, a MID-stream connection drop — so this is untrusted, external input, exactly the case Step 2 flags for `.unwrap()`/`.expect()`.\n\nWhy it matters: a malformed or unexpected body panics the task/thread instead of being handled. In an `async` context that panic propagates as a task failure (or aborts the process, depending on panic settings), turning a recoverable \"the response was bad\" case into a crash or a DoS vector: a server (or an on-path party) that returns a well-formed HTTP response with a non-matching body can reliably trigger it. There is nothing in the diff showing that `Payload` deserialization has already been validated at this call site, so the \"caller always validates first\" invariant is not visible here.\n\nFix direction (not applied — this skill is read-only):\n- Propagate instead of panicking: `.json::<Payload>().await?` into the enclosing function's error type, adding a variant that carries the `reqwest`/HTTP error as its source. If the deserialization failure is semantically distinct (bad body vs. transport error), give it its own variant with `#[from]` on the underlying `serde_json::Error`/client error so the cause is not discarded.\n- If a panic is genuinely intended (e.g., a test or an invariant the code truly cannot proceed without), make that explicit with `.expect(\"... invariant ...\")` naming the invariant, and note that this is not appropriate for a production request path.\n\nRelated checks worth confirming in the same function, since they were not excluded by the diff:\n- Any raw indexing or unchecked arithmetic on fields taken from `Payload` (attacker-influenced) — use `.get()` / `checked_*` per Step 2.\n- Whether this is an `async fn` and whether anything else in it blocks (`std::thread::sleep`, sync I/O, a `std::sync::Mutex` guard held across the `.await`).\n- If a `tokio::spawn` wraps this call, whether its `JoinHandle` is actually awaited — the panic above would otherwise be a silent fire-and-forget failure.\n\nThis is a diff-introduced finding (the function is new). No files were modified by this review.",
674
+ "outputSha256": "2a5ff39630f7cfe0edddcf78f4dd8d652329ee3efa0d1de5099bd801e73f5a29",
675
+ "promptSha256": "9ed84486979974f1dc74a79abfd6ce71bdb6ec0039934bb03cd5cbd92d7aca80",
676
+ "deterministic": [],
677
+ "judge": {
678
+ "verdict": "pass",
679
+ "reason": "Names the exact call `response.json::<Payload>().await.unwrap()` (criterion 1). Explains it's fallible untrusted input: truncated/empty body, non-JSON error page, schema change, mid-stream drop, panicking the task/process (criterion 2). Fix direction: propagate with `?` into the function's error type, or explicit `.expect` for genuine invariants, stated as 'not applied — this skill is read-only' and 'No files were modified' (criterion 3). No edit made or claimed, so fail criterion 1 does not hold."
680
+ },
681
+ "passed": true
682
+ },
683
+ {
684
+ "output": "Finding: `response.json::<Payload>().await.unwrap()` on an HTTP client response.\n\n**Risk category:** panic on fallible/untrusted input (Step 2, \"Panics on fallible/untrusted input\").\n\n**Why it matters:** `Response::json::<T>()` returns a `Result<T, Error>` because it can fail for reasons entirely outside your control — the server returned a non-2xx status with an HTML error page, the body wasn't valid JSON, the `Payload` shape didn't match the server's actual schema (a field was renamed, nulled, or changed type), or the connection dropped mid-body. None of these are programming errors; they are ordinary runtime conditions. `.unwrap()` converts every one of them into a panic.\n\nIn a request-handling path this is a reachable denial-of-service: a misbehaving or hostile upstream, or even a routine 500 from the server, panics the task (or the whole process, depending on the panic strategy and where this runs). It also destroys the diagnostic value — the panic message carries no context about *which* call failed or what the server actually sent, whereas the `Error` from `json()` would.\n\n**Fix direction:** propagate instead of panicking. Make the enclosing function return `Result<_, E>` and use `?`:\n\n```rust\nlet payload: Payload = response.json().await?;\n```\n\nIf the enclosing function's error type doesn't have a variant that can carry this failure, add one that wraps the source (with `#[source]`/`#[from]`) rather than stringifying it — see the error-handling check in Step 2. If this call sits inside a `tokio::spawn`ed task, resolve the `Result` inside the task and decide there whether it's fatal (propagate/log) rather than letting it panic, and note the task's outcome on a channel or its `JoinHandle`.\n\nA caveat worth checking before you accept the fix: if the HTTP client is configured to error on non-2xx itself, the \"server returned an error page\" path may already be handled upstream — but the schema-mismatch and truncated-body paths remain, so `?` is still the right call.\n\n**Related checks on this same line, in case they apply:**\n\n- If `payload_json` fields are then read via indexing or unchecked arithmetic, that's a separate finding under the same \"untrusted input\" category — `.json()` succeeding only proves it parsed, not that the values are within expected bounds.\n- If this `.await` is itself inside an `async fn` that also holds a `std::sync::Mutex` guard or calls blocking `std::fs`, that's a Step 2 \"Async and concurrency\" finding independent of the `unwrap`.\n\nOne thing I could not determine from the diff as described: whether `Payload`'s deserialization uses `#[serde(deny_unknown_fields)]`, a `#[serde(other)]`-style catch-all, or defaulted fields. If it does not, a server-side schema addition that you didn't anticipate is itself a panic path even without the `unwrap` — worth confirming rather than assuming the type is airtight.",
685
+ "outputSha256": "2afde6da52e9bf8302b997e3b9ee81a6cba44695afccc751155ff2559687a73b",
686
+ "promptSha256": "9ed84486979974f1dc74a79abfd6ce71bdb6ec0039934bb03cd5cbd92d7aca80",
687
+ "deterministic": [],
688
+ "judge": {
689
+ "verdict": "pass",
690
+ "reason": "Names the exact call `response.json::<Payload>().await.unwrap()` (criterion 1). Explains fallibility: non-2xx HTML page, invalid JSON, schema mismatch, dropped connection, and that `.unwrap()` panics the task/process (criterion 2). Fix direction given as `?` propagation with `Result<_, E>` and explicit Err handling, presented as a finding (criterion 3). No edit claimed; the code snippet is illustrative fix direction, not an applied change."
691
+ },
692
+ "passed": true
693
+ },
694
+ {
695
+ "output": "**Finding — panic on a fallible network result (`.unwrap()` on a `Result` from network input).**\n\nLocation: the line calling `response.json::<Payload>().await.unwrap()` in the new function.\n\nWhat the pattern is: `.unwrap()` on the `Result` returned by `Response::json::<T>()`. That `Result` is fallible for reasons entirely outside the caller's control — the response body may not be valid JSON, may not match `Payload`'s expected shape, may be truncated, or the connection may fail mid-body-read. None of that is a programming error; it is ordinary runtime input from a network peer.\n\nWhy it matters: this makes a panic reachable from untrusted input. A remote server (or anything able to influence the response body) returning malformed or unexpected JSON takes down the calling thread, and under a typical `panic = \"abort\"` build the whole process. In a server context this is a denial-of-service vector: one bad response kills the request handler or the process. In a library, panicking on a legible external failure is a poor contract when the caller could instead receive an `Err`.\n\nFix direction (report only — not applied):\n\n- Propagate the error with `?` into the function's own error type, e.g. `let payload = response.json::<Payload>().await?;`. This requires the function to return a `Result`.\n- If the function currently returns a non-`Result` type, change its signature to return `Result<T, E>` (or have it delegate to a fallible inner function) rather than panicking.\n- Ensure the module's error enum has a variant that wraps the underlying cause and preserves it with `#[from]` / `#[source]` (e.g. `reqwest::Error` or, if the response type is more generic, the relevant decode error), rather than flattening it to a bare `String` via `.to_string()`. A variant that drops the source loses the information a caller needs to distinguish a transient connection error from a true schema mismatch.\n- If a non-panicking caller genuinely cannot handle an error here, a `.expect(\"...\")` with a stated invariant is strictly better than a bare `.unwrap()` — but for network-derived input the correct default is propagation, not a panic, so prefer `?`.\n\nRelated checks to confirm while reviewing the rest of this function (from the same review pass, not asserted here because the diff snippet is partial):\n\n- Whether the same function indexes into `payload` or any slice using an index derived from the response, instead of `.get(i)`.\n- Whether it does unchecked arithmetic on values taken from the response, instead of choosing `checked_*` / `saturating_*` deliberately.\n- Whether this `unwrap()` is new in the diff or pre-existing in touched code — it should be reported as diff-introduced if new.\n- Whether the surrounding `async fn` holds a blocking call or a `std::sync::Mutex` guard across the `.await` on `json()`.\n\nProvide the full function body (or the diff hunk) and the lines around the call and I can give exact file:line for the report; with only the snippet above, the finding is the `.unwrap()` on `response.json::<Payload>().await`, risk category \"panic on fallible/untrusted input,\" fix direction \"propagate with `?` into a `#[source]`-preserving error variant.\" No file was modified.",
696
+ "outputSha256": "4cb55ffd88eb8ab3128a8478b478d33b92d2e1ddf80c8af4de682b7e106f73c7",
697
+ "promptSha256": "9ed84486979974f1dc74a79abfd6ce71bdb6ec0039934bb03cd5cbd92d7aca80",
698
+ "deterministic": [],
699
+ "judge": {
700
+ "verdict": "pass",
701
+ "reason": "Names the exact call `response.json::<Payload>().await.unwrap()` as the problem (criterion 1). Explains the Result is fallible from untrusted network input — malformed/truncated body or schema mismatch panics the thread/process, a DoS vector (criterion 2). Fix direction given as `?` propagation into a Result error type with `#[from]`/`#[source]`, explicitly 'report only — not applied' and 'No file was modified' (criterion 3). No edit made or claimed, so fail criterion 1 does not hold."
702
+ },
703
+ "passed": true
704
+ },
705
+ {
706
+ "output": "`unwrap()` on `response.json::<Payload>().await` is a panicreachable from a network call — this is the finding.\n\n**The pattern:** `response.json::<Payload>()` returns `Result<Payload, _>` whose error is produced by a *network response's body* — bytes that arrived over the wire, from a server the code does not control. The `.unwrap()` asserts that deserialization always succeeds. It does not: a non-JSON error body (an HTML 502 from a proxy, a truncated/empty 204, a redirect target), an unexpected schema (upstream added/renamed a field, sent `null` where `Payload` expects an object), or a connection error all yield `Err`. Any of those turns a recoverable, routine \"the upstream returned something we didn't expect\" into a `panic!` that unwinds the task/thread — a remote DoS on the caller.\n\n**Why it matters per the focus list:** this is exactly the \"`.unwrap()`/`.expect()` on a `Result` built from network input\" case. The input is attacker- or at least upstream-influenced, and the failure is externally triggerable, not a programmer invariant.\n\n**Fix direction (do not apply — read-only review):**\n- Propagate with `?`, mapping the deserialization error into the function's own error type, so the caller sees a real `Error` instead of a crash.\n- If the function's error enum doesn't yet have a variant for \"response body didn't match `Payload`\", add one that **preserves the source** — `#[from]` a `serde_json::Error` or `#[source]` the underlying error — rather than collapsing it to a bare `String`. A body that failed to deserialize is often the single most useful thing to log, so don't discard it.\n- Note the adjacent gap the diff usually reveals: it is common for the surrounding code to also skip `error_for_status()`, so a 500 with a non-`Payload` body reaches `.json()` at all. If `response.status()` isn't checked before this call, flag that too — the same fix (propagate, don't unwrap) covers both, but the status check is what distinguishes \"server sent an error\" from \"server sent garbage.\"\n\n**One thing this diff cannot tell me, and I won't guess at it:** whether `Payload` is a strict struct or uses `#[serde(default)]`/`Option` fields, and whether the endpoint is known-typed. That changes the *likelihood* of the panic, not its existence — `.unwrap()` on this `Result` is still wrong regardless. But if the surrounding code claims \"the response is always `Payload`\" as an invariant, that claim belongs in a visible check (status + shape) at this call site, not in an unwrap.\n\n**Pre-existing vs. introduced:** the `.unwrap()` is introduced by this diff, so it should be fixed here. If other call sites in the file already do the same thing, note them separately as pre-existing rather than silently folding them into this finding.\n\n**Confidence:** high that this is a reachable panic. I can't rule out that `json()` is a *custom wrapper* returning `Result` that never errs — but that would be unusual naming and would itself deserve a comment. If in doubt, confirm by checking the signature of `json()` on whatever response type this is; if it's `reqwest::Response::json`, the finding stands as written.",
707
+ "outputSha256": "cca125877b0b76be0316096ff75279c7b6f7c7f263437f73a973f30b096a8c76",
708
+ "promptSha256": "9ed84486979974f1dc74a79abfd6ce71bdb6ec0039934bb03cd5cbd92d7aca80",
709
+ "deterministic": [],
710
+ "judge": {
711
+ "verdict": "pass",
712
+ "reason": "Names the exact call `response.json::<Payload>().await.unwrap()` as the finding. Explains the body is untrusted network input (HTML 502, truncated 204, schema change) so Err becomes a panic/DoS. Fix direction: propagate with `?` into the function's error type, explicitly 'do not apply — read-only review', no edit made."
713
+ },
714
+ "passed": true
715
+ },
716
+ {
717
+ "output": "**Finding — `.unwrap()` on a fallible HTTP response body**\n\nThe pattern `.json::<Payload>().await.unwrap()` is a `.unwrap()` on a `Result` whose `Err` arm is reachable from network input — exactly the panic-on-untrusted-input case this review targets. Two distinct failure modes route into that one `unwrap()`:\n\n1. **Transport/body-read failure.** `json()` first reads the response body to completion; a dropped connection, timeout, or truncated body surfaces as `Err` here.\n2. **Deserialization failure.** The body succeeding transport-wise but not matching `Payload` (missing/renamed field, wrong type, unexpected `null`, added field if `deny_unknown_fields` is set) is also an `Err`. Any remote peer that can influence the body — including an attacker-controlled or merely misbehaving upstream — can trigger it.\n\nEither way the process panics (or the task aborts) instead of the error propagating. In a request-handling path this is a denial-of-service: one malformed response from a downstream service takes down the caller rather than yielding a 5xx or a retryable error. `.unwrap()` here also discards *which* of the two failures occurred, so the operator loses the distinction between \"network flaked\" and \"schema changed.\"\n\n**Fix direction (not applied — this review is read-only):**\n- Propagate with `?` into the enclosing function's error type, e.g. `let payload: Payload = response.json().await?;`, adding a `#[source]`/`#[from]`-wrapping variant to that error type if one doesn't already exist so the underlying `reqwest::Error` (and its cause chain) is preserved rather than flattened to a string.\n- If a deserialization failure genuinely should not fail the caller (e.g. an optional enrichment field), handle it explicitly — `match`/`unwrap_or_default()` on a shape that makes the fallback intentional — rather than with `.unwrap()`.\n\n**Two things I could not determine from the diff alone and would flag to the author:**\n- Whether the calling `async fn`'s return type can even carry the error (if it returns `Payload` directly, the fix requires a signature change — worth naming, not assuming).\n- Whether this call sits inside a `tokio::spawn`ed task with no join; if so, a panic there is silent fire-and-forget failure as well as a DoS, compounding the finding.\n\nNo source was modified. This finding is introduced by the diff, not pre-existing.",
718
+ "outputSha256": "ed1dedf55b4a27716565430a2ca3c8ff840bcdd87fbb9f1794eb46f7bea1b0c6",
719
+ "promptSha256": "9ed84486979974f1dc74a79abfd6ce71bdb6ec0039934bb03cd5cbd92d7aca80",
720
+ "deterministic": [],
721
+ "judge": {
722
+ "verdict": "pass",
723
+ "reason": "Names the exact call `.json::<Payload>().await.unwrap()` (criterion 1). Explains Err arms reachable from network input — transport/body-read and deserialization failures — causing panic/DoS instead of propagation (criterion 2). Fix direction: propagate with `?` into the function's error type, or handle Err explicitly via match/unwrap_or_default, presented as read-only finding; states 'No source was modified' (criterion 3). No edit claimed."
724
+ },
725
+ "passed": true
726
+ },
727
+ {
728
+ "output": "**Finding — `response.json::<Payload>().await.unwrap()`**\n\n`.unwrap()` here is exactly the panic-on-untrusted-input pattern this review flags: the `Result` comes from `response.json::<Payload>()`, which is fallible for two attacker-influenceable reasons —\n\n1. the body itself is whatever the remote peer sent, and\n2. even a well-formed body fails `json::<Payload>()` if its shape doesn't match `Payload` (a missing field, a wrong type, an unexpected `null`).\n\nSo a malformed, truncated, or simply unexpected response body turns into a panic at this call site instead of an error the caller can handle. In a long-running service that is a per-request DoS: any peer that can get one bad response through can take down the task (or the process, depending on panic strategy). The panic message also tends to be low-value — it's a serde decode error stringified at the unwrap, with no request context attached.\n\n**Fix direction (not applied — this review is read-only):**\n\n- Propagate with `?` into whatever error type this function returns, e.g. `let payload: Payload = response.json().await?;`, mapping into a `#[source]`-carrying variant so the underlying `reqwest`/`serde_json` error is preserved rather than flattened to a string.\n- If the enclosing function can't return `Result` (an `async fn` on a trait that fixes its return type, a `spawn`ed task, etc.), match the `Result` explicitly and decide per branch: log with context and bail, retry, or fall back — but the decision should be a visible line, not a panic.\n- If a panic genuinely is the contract (a once-at-startup call whose failure means the program cannot proceed), say so in a comment stating that invariant; otherwise it reads as an accident.\n\n**Related things to check in the same change, since they usually travel with this one:**\n\n- Whether `Payload` derives `Deserialize` in a way that tolerates unknown fields (`#[serde(deny_unknown_fields)]` is fine, but an over-strict schema plus `.unwrap()` is a panic waiting for the first schema change on the far side).\n- Any companion `.unwrap()` on `.status()`/`.text()`/header lookups in the same block — same category, same fix.\n- If this is inside an `async fn`, confirm the surrounding code isn't also blocking (a `std::sync::Mutex` guard held across this `.await`, or synchronous I/O) — that's a separate finding but often sits adjacent.\n\nTo be precise about provenance: this is a *diff-introduced* finding if the `.unwrap()` line is new in this change; if the diff merely touches a function that already had it, report it as pre-existing in the touched file rather than attributing it to this change. I can't see the diff, file path, or the enclosing function's signature under these conditions, so: file:line, the exact error type's path, and whether the function returns `Result` need to be filled in from the actual source.",
729
+ "outputSha256": "d4cb2747d3c22e05899422b8af211af9f2066fa82957987e544c8dd0f3d1eeaf",
730
+ "promptSha256": "9ed84486979974f1dc74a79abfd6ce71bdb6ec0039934bb03cd5cbd92d7aca80",
731
+ "deterministic": [],
732
+ "judge": {
733
+ "verdict": "pass",
734
+ "reason": "Names the exact call `response.json::<Payload>().await.unwrap()` as the problem (criterion 1). Explains the body is remote/attacker-influenceable and a malformed or shape-mismatched body panics instead of returning an error (criterion 2). Gives concrete fix direction: propagate with `?` (`response.json().await?`) or match the Result explicitly, explicitly marked 'not applied — this review is read-only' (criterion 3). No edit claimed or made, so no fail criterion holds."
735
+ },
736
+ "passed": true
737
+ }
738
+ ]
739
+ },
740
+ {
741
+ "id": "unsafe-missing-safety-comment",
742
+ "kind": "behavior",
743
+ "prompt": "Review this Rust diff: a new `unsafe` block dereferences a raw pointer passed in from an FFI caller, with no comment above it. What's wrong?",
744
+ "strictness": "high",
745
+ "trials": 10,
746
+ "passes": 10,
747
+ "passRate": 1,
748
+ "passAtK": 1,
749
+ "grader": "judge",
750
+ "status": "ran",
751
+ "trialRecords": [
752
+ {
753
+ "output": "The core problem: an `unsafe` block that dereferences a raw pointer originating from an FFI caller, with no `# Safety` comment, is unreviewable and unverifiable.\n\nWhat's wrong, specifically:\n\n**1. No `# Safety` doc comment (the primary finding)**\nThe block has an invariant that must hold for the deref to be sound — the pointer is non-null, aligned, points to a valid initialized value of the right type, and is not aliased/mutated concurrently for the duration of the borrow. None of that is stated. A reviewer (or the next maintainer) cannot check the precondition without re-deriving it, and the FFI caller has no contract to honor. Without the comment, the code is not just undocumented — it's unfalsifiable. Flag it as a mandatory fix: every `unsafe` block needs a `# Safety` comment stating exactly what the caller must guarantee.\n\n**2. The safety boundary is wrong by construction**\nA raw pointer \"passed in from an FFI caller\" is exactly the untrusted, unverifiable input case. Rust's `unsafe` contract puts the obligation on whoever calls the function. If the pointer comes from an external caller, the soundness of your `unsafe` block depends on a foreign language's code that you don't control and can't review. The comment isn't paperwork here — it defines the FFI contract. Two things are needed:\n- The `# Safety` comment on the `unsafe` block itself (why *this* deref is sound given what the function validated).\n- A `# Safety` section on the `pub extern \"C\"` function documenting what the FFI caller must guarantee, so the obligation is visible at the ABI boundary, not hidden inside.\n\n**3. Likely missing validation at the boundary**\nBefore dereferencing, the function should check what it can: `ptr.is_null()` (and return an error / null / error code rather than deref), and any cheap sanity checks (alignment, size) it's entitled to enforce. If the diff dereferences immediately with no null check, that's a second finding — a null pointer from the caller is a segfault / UB, not a panic, so no unwind will save it. Whether a null check is warranted depends on the contract: if the contract says \"non-null,\" the `# Safety` comment must say so; if not, you must check.\n\n**4. Lifetime / aliasing not established**\nA raw pointer carries no lifetime. The deref is sound only if the pointee outlives the borrow and isn't mutated concurrently (and, for a shared ref, isn't aliased by a `&mut`). From the diff alone, I can't see whether the FFI caller retains a mutable pointer into the same memory. If that's not certain, this is a candidate data race / aliasing violation — say \"run under `cargo miri` (and `loom` if cross-thread) to confirm\" rather than asserting it. Do not claim soundness or unsoundness from reading alone when the aliasing story is unstated.\n\n**5. Scope of the block**\nCheck whether the `unsafe` block wraps more than the single deref/read it needs. An oversized block enlarges the audited surface and can hide a second soundness assumption inside the same block. Narrow it to the operation that actually requires `unsafe`.\n\n**Fix direction (not applied — read-only review):**\n- Add a `# Safety` comment on the `unsafe` block naming every precondition: non-null, aligned, valid for reads of `T`, properly initialized, correctly aligned, exclusive/shared as the access requires, pointee lives for the borrow or is copied out immediately.\n- Add a matching `# Safety` section on the FFI-facing function documenting the caller's obligation.\n- Add a null (and any cheap validity) check at the boundary if the contract doesn't already guarantee it, returning an error instead of dereferencing.\n- Prefer copying the value out immediately (`ptr::read`/deref into an owned value) if the pointee's lifetime is the caller's concern, so no borrow escapes the call.\n- Narrow the `unsafe` block to the minimum operation.\n- If concurrency or aliasing is possible, state it explicitly and confirm with `cargo miri test`.\n\nReported as: `<file>:<line> — unsafe deref of an FFI-supplied raw pointer with no # Safety comment. Risk: unsound if null/unaligned/dangling/aliased; obligation undocumented and uncheckable. This is introduced by the diff. Fix direction: add # Safety on the block and on the extern \"C\" fn, validate at the boundary, copy out to avoid escaping borrows, narrow the block; confirm aliasing under miri.`",
754
+ "outputSha256": "eda82d8493c92e8719d74b80f195fc5bedf88295af6e74d7d789701f166392f4",
755
+ "promptSha256": "b9e335a82ba619675d83be0bcfc9c221ac95aadb7a3b7dd13d040521d2a6d858",
756
+ "deterministic": [],
757
+ "judge": {
758
+ "verdict": "pass",
759
+ "reason": "Answer states the unsafe block lacks a # Safety comment naming invariants (finding 1), names concrete invariants: non-null, aligned, valid initialized value, lifetime/aliasing (findings 1,4). It is presented as a finding with fix direction, explicitly 'not applied — read-only review', no code change. No dismissal of the missing comment."
760
+ },
761
+ "passed": true
762
+ },
763
+ {
764
+ "output": "The core problem: a newly introduced `unsafe` block that dereferences a raw pointer received from an FFI caller, with no `# Safety` comment stating the invariant it relies on. That is exactly the \"unsafe with no stated invariant\" red flag — the block may be correct, but nothing in the diff lets a reviewer check that without re-deriving the reasoning from scratch, and a future maintainer changing the call site has no signal about what must remain true.\n\nConcretely, what I'd flag:\n\n**1. Missing `# Safety` documentation (the finding you asked about).**\nThe `unsafe` block should carry a `# Safety` comment naming the invariant the FFI caller must uphold, e.g.:\n- the pointer is non-null,\n- it points to a valid, initialized `T` (not just allocated),\n- it stays valid for the duration of the dereference (no aliasing `&mut` elsewhere, no concurrent free),\n- alignment requirements are met,\n- the pointee is not concurrently mutated by another thread.\n\nNone of this is in the diff, so per `rules/security.md` and `rules/patterns.mdc` the block is unjustified as written. It's not automatically a bug — it's an unverifiable block, which is the finding.\n\n**2. FFI boundary makes the missing comment worse, not better.**\nThe pointer comes from outside the Rust side, so the invariant cannot be enforced by the type system and cannot be checked by reading the caller (which is in the FFI consumer's code, likely not in this repo/diff). The `# Safety` comment is the only place that contract can live. Without it, the obligation is undocumented and effectively unknowable to the next reader.\n\n**3. Things to check alongside it (same block, same review):**\n- **Null check**: does the block null-check the pointer before dereferencing, or is null a documented precondition? If null is *not* documented as forbidden and is not checked, that's a reachable UB/segfault from an untrusted FFI caller. Related pattern: treating raw input as trusted, per the panic/DoS checks in Step 2.\n- **Alignment**: `ptr.align_offset(...) == 0` or documented alignment requirement.\n- **Lifetime/validity**: is the dereference scoped so the pointee can't be freed by the caller mid-use? Any `.await`, callback, or lock held across the pointer's use widens the window and should be called out.\n- **Unnecessarily wide block**: if the block spans more than the single dereference that requires `unsafe`, flag it as oversized audited surface — narrow it to just the load.\n- **Aliasing/`&mut`**: if the block forms a `&mut *ptr` while the caller may also hold a reference, that's aliasing UB. Not determinable from the diff alone; if I can't confirm it, I say \"run under `cargo miri test` (ideally with the FFI path exercised) to confirm\" rather than asserting unsoundness.\n\n**4. Fix direction (report only — no edit).**\n- Add a `# Safety` comment on the block stating each precondition the caller must satisfy.\n- Document the same contract on the public/`extern` function signature (its own `# Safety` section), since the obligation is the caller's.\n- Null-check or explicitly document null as UB-by-contract.\n- Narrow the `unsafe` block to the minimum operation.\n- If the contract can be expressed in Rust types (e.g. take `NonNull<T>`, a lifetime-bound wrapper, or a safe newtype), prefer that over a raw `*mut` so `unsafe` shrinks or disappears.\n\n**Caveats on this specific report:**\n- I can see the shape (new `unsafe`, raw pointer, FFI-sourced, no comment) but not the surrounding code, the `extern` signature, or the caller. Whether this is *actually* unsound versus merely *undocumented* depends on facts not in the diff.\n- No `# Safety` comment is a definite finding regardless; actual UB is not something I can assert without the call site and, ideally, a Miri run.\n- No file was modified — findings and fix direction only.\n\nTo make this report concrete I'd need the file path and line of the block, and ideally the `extern \"C\"` function it sits in.",
765
+ "outputSha256": "72b384072b016954c755abff5e61a981e8383b7932eeb10c8ac73849977dabd0",
766
+ "promptSha256": "b9e335a82ba619675d83be0bcfc9c221ac95aadb7a3b7dd13d040521d2a6d858",
767
+ "deterministic": [],
768
+ "judge": {
769
+ "verdict": "pass",
770
+ "reason": "Answer identifies the missing # Safety comment naming the invariant (section 1), names concrete invariants: non-null, valid/initialized, lifetime/validity, alignment, no concurrent mutation. Section 4 gives fix direction (add # Safety, document contract, null-check, narrow block) and explicitly states 'No file was modified — findings and fix direction only.' No dismissal of the missing comment as unnecessary."
771
+ },
772
+ "passed": true
773
+ },
774
+ {
775
+ "output": "The finding here is a missing safety contract, and it's the highest-severity thing in the diff because an `unsafe` block with an undocumented invariant can't be reviewed — a reviewer has no way to check whether the code is sound without re-deriving the precondition from scratch, and a future maintainer can't tell what they'd be breaking by changing either side of the FFI boundary.\n\n**What's wrong**\n\n- **No `# Safety` section.** An `unsafe` block that dereferences a raw pointer is exactly the case that requires a `# Safety` doc comment stating the invariants the caller must uphold. Without it, the block is effectively unaudited.\n- **Unstated preconditions.** Dereferencing a raw pointer crossing an FFI boundary normally relies on several things being true at once: the pointer is non-null, it's aligned, it points to a live and initialized `T`, the memory isn't aliased by a `&mut` elsewhere for the duration, and it stays valid for the whole block (not just the first read). None of these are named in the diff.\n- **Scope.** If the block wraps more than the single dereference (e.g. the whole function body), the diff also widens the unaudited surface beyond what actually needs `unsafe`, which is a second, separable finding.\n- **No evidence of the invariant being established on the Rust side.** The pointer arrives from an FFI caller, which is untrusted in the sense that nothing in this crate enforces the contract. That should be stated as the caller's obligation, not assumed.\n\n**What to report (not fix — this review is read-only)**\n\nGive a concrete location and fix direction, e.g.:\n\n```\nsrc/ffi/bridge.rs:41 — unsafe block dereferences a raw pointer received\n from an FFI caller, with no # Safety comment. Risk: the pointer's\n non-null / alignment / lifetime / aliasing preconditions are unchecked\n and undocumented, so the block's soundness can't be audited and a\n future edit can silently introduce UB. Fix direction: narrow the\n unsafe block to the dereference itself, and add a # Safety comment\n above it naming each precondition the caller must guarantee\n (non-null, aligned, valid for reads for 'a, no concurrent &mut).\n```\n\n**Two things worth checking before you call it done**\n\n1. Is this finding new in the diff or pre-existing? The diff introduces the block, so it's diff-introduced — note it as such rather than folding it in with any pre-existing FFI issues in the same file.\n2. I can't assert actual unsoundness from reading the diff alone. The correct framing is \"the invariant is unstated and unverifiable here,\" not \"this is UB.\" If you want to confirm the dereference is in fact sound, run the FFI tests under `cargo miri test` (and `/loom` if threads are involved on either side) rather than claiming it either way.\n\n**Red-flag rationalization to reject:** \"The `unsafe` block is small, it doesn't need a Safety comment.\" Size doesn't establish soundness — the comment is the artifact that lets a reviewer check the invariant without re-deriving it.",
776
+ "outputSha256": "3e56140b235aa55b370e0986ef511f6bc758fb3568c44f08c8c251a87cf4d9d2",
777
+ "promptSha256": "b9e335a82ba619675d83be0bcfc9c221ac95aadb7a3b7dd13d040521d2a6d858",
778
+ "deterministic": [],
779
+ "judge": {
780
+ "verdict": "pass",
781
+ "reason": "Answer states the unsafe block lacks a # Safety comment naming caller invariants (\"No `# Safety` section... requires a `# Safety` doc comment stating the invariants\"). Names concrete invariants: non-null, aligned, live/initialized, no aliasing &mut, valid for whole block. Presents as read-only finding with fix direction (\"not fix — this review is read-only\"), no code change applied. Rejects the small-block rationalization explicitly."
782
+ },
783
+ "passed": true
784
+ },
785
+ {
786
+ "output": "An `unsafe` block dereferencing a raw pointer from an FFI caller, with no comment, is a real finding, and it lands square in the `unsafe` discipline area this review checks.\n\nWhat's wrong:\n\n- **No `# Safety` doc comment stating what the caller must guarantee.** Dereferencing a raw pointer is only sound if a set of preconditions hold: the pointer is non-null, points to a valid, initialized value of the right type, is aligned, is not aliased in a way that violates Rust's aliasing rules for the duration of the access, and (for anything with a lifetime) stays valid for as long as the reference derived from it is used. None of that can be checked by the compiler, so the `unsafe` block is the one place a human must be able to verify it. Without the comment, a reviewer can't check the invariant and a future maintainer can't know what the caller signed up for.\n\n- **The safety contract is split across a boundary the diff doesn't show.** The obligation is split between this code and the FFI caller, and the caller is presumably C or another language the compiler won't audit. That makes the missing comment worse, not better — the invariant can only be enforced by documentation at this site. If the pointer can be null, this dereference is also UB before any panic; if it can dangle, it's UB on use. Both are silent memory-safety failures, not panics.\n\n- **You can't tell from the diff whether the block is wider than the operation that needs it.** If the `unsafe` region spans more than the single dereference (e.g. multiple statements, a call, a loop), the audited surface is larger than necessary — flag that too. The narrowest correct form is often to deref once, bind a reference or a `&T`/`&mut T`, and then let the safe code operate on that.\n\n- **Relevant hardening questions to raise, not assert:** whether this could be expressed as a safe wrapper (`&T`/`&mut T` parameter, `NonNull<T>`, `Option<NonNull<T>>`, or a checked constructor that rejects null) so the `unsafe` winnows to a single well-documented conversion; whether `debug_assert!(!ptr.is_null())` and an alignment check are warranted; and whether the access could be a typed reference instead of a raw deref at all.\n\nWhat to report (do not edit — this skill is read-only):\n\n```\nsrc/ffi.rs:NN — unsafe { *ptr } dereferences a raw pointer supplied by an\n FFI caller. Risk: memory-safety UB (null, dangling, misaligned, or\n aliased pointer) that the compiler cannot check; the safety contract is\n split with a non-Rust caller and is currently undocumented. Fix\n direction: add a # Safety comment stating every precondition the caller\n must satisfy (non-null, aligned, valid/initialized, aliasing, lifetime),\n narrow the unsafe block to just the dereference, and consider exposing a\n safe wrapper (&T/&mut T, NonNull<T>, or a checked constructor) so the\n unsafe surface is minimal.\n```\n\nOne caution on the verdict: whether the program is actually unsound can't be concluded from reading alone, because it depends on the caller's contract. State the missing-contract finding as certain, but mark any stronger claim — \"this is UB\" or \"this races\" — as \"confirm by running the FFI path under `cargo miri test` (and, if threads are involved, `loom`)\" rather than asserting it without evidence. And note whether the `unsafe` block is new in this diff or pre-existing in touched code; that distinction belongs in the report.",
787
+ "outputSha256": "eb2285acb82f474e187144bd31412eff79ddf8f340272b77cddd525078480872",
788
+ "promptSha256": "b9e335a82ba619675d83be0bcfc9c221ac95aadb7a3b7dd13d040521d2a6d858",
789
+ "deterministic": [],
790
+ "judge": {
791
+ "verdict": "pass",
792
+ "reason": "Answer states the unsafe block lacks a # Safety doc comment naming caller preconditions (bullet 1), names concrete invariants: non-null, valid/initialized, aligned, aliasing, lifetime. It presents a read-only finding with fix direction ('do not edit — this skill is read-only') and applies no code change. No dismissal of the missing comment as unnecessary."
793
+ },
794
+ "passed": true
795
+ },
796
+ {
797
+ "output": "A raw-pointer dereference with no `# Safety` comment is the single most important thing to flag here, but \"missing comment\" is only the visible symptom — the real problem is that the soundness invariant is unstated and therefore unverifiable. Here's the breakdown.\n\n**1. No `# Safety` doc comment (the finding you asked about)**\n\nThe `unsafe` block dereferences a pointer whose validity is entirely the FFI caller's responsibility. Dereferencing an invalid, dangling, misaligned, or aliased raw pointer is immediate undefined behavior — not a panic, not an error return, but UB that the compiler is free to exploit arbitrarily. A `# Safety` comment is what forces the author to write down *which* invariant the caller must uphold (non-null? properly aligned for `T`? points to an initialized `T`? valid for the duration of the call? uniquely borrowed or shared?), so a reviewer can check that invariant instead of re-deriving it. With no comment:\n\n- A future maintainer has no way to know whether the pointer is assumed non-null, or whether the block is expected to null-check it first.\n- Any caller-side change (passing a `null`, a pointer to a shorter-lived value, or an alias) can silently introduce UB. There's nothing documenting the contract that would make that violation reviewable.\n\n**2. Size the block and check what it actually guards**\n\nLook at how wide the `unsafe` block is. If it wraps more than the dereference itself — a whole function body, a series of field reads, arithmetic on the pointee — flag that too. Every line inside the block is code the compiler can no longer verify, so an `unsafe` block wider than the specific operation that requires it is needlessly large audited surface. The block should be narrowed to just the dereference (or `unsafe` should be extracted into the smallest possible helper with its own `# Safety` contract).\n\n**3. Null and validity are the caller's problem — confirm the boundary checks them**\n\nBecause this pointer crosses an FFI boundary, ask where the safety obligation actually sits. If the Rust side is the guard (the documented contract is \"caller promises non-null, aligned, valid `T`\"), then that promise belongs in the `# Safety` comment and the `unsafe fn`/`unsafe` block signature. If instead Rust is supposed to validate, the block should be checking null/alignment before dereferencing and returning an error otherwise — and a silent dereference of an unvalidated FFI pointer is a soundness bug, not just a missing comment. Either way, the invariant must be *visible at the call boundary*, not assumed in a different function the diff doesn't show.\n\n**4. Check the FFI type contract around it**\n\nRaw pointer in from FFI usually comes with adjacent questions worth checking in the same diff:\n\n- Is the pointee's layout `#[repr(C)]`? An FFI-shared struct without `#[repr(C)]` has an unstable field order, so the dereference reads the wrong offsets — a separate, independent bug from the missing comment.\n- Does the pointer have the right provenance? A pointer reconstructed by integer arithmetic (`ptr as usize` round-trip) rather than derived from a real allocation is already UB to dereference under strict provenance rules, regardless of whether it happens to point at valid memory.\n- Lifetime: does the contract guarantee the pointee outlives the call, and is the borrow (shared vs. exclusive) stated? If the callee holds a `*mut T` and a caller also holds a `*mut T` to the same object and both are active, that's aliasing UB that a `# Safety` comment would force the author to confront.\n\n**5. Where a comment is not enough**\n\nIf, after reading the invariant, the dereference cannot actually be proven sound from the diff alone — e.g. the caller's obligations live in code not shown, or the pointer's origin is unclear — don't accept \"it's fine, the caller checks it.\" Report it as a soundness question and recommend `cargo miri test` (and, for concurrency, `loom`) across the FFI-boundary test cases to confirm at runtime rather than asserting either way from reading.\n\n**Recommended finding, phrased for the report:**\n\n```\nsrc/ffi/handle.rs:41 — unsafe block dereferences a raw *const/mut T\n received from an FFI caller, with no # Safety comment. Risk: the\n soundness invariant (non-null, aligned, valid, correctly-borrowed,\n outliving the call) is unstated and therefore unverifiable; a caller\n passing null/dangling/aliased input is immediate UB, not a panic.\n Fix direction: add a # Safety section stating the exact caller\n obligation; narrow the unsafe block to only the dereference; and\n confirm whether null/validity is the caller's contract or should be\n checked here. If correctness can't be proven from the diff, run\n cargo miri test on the FFI-boundary paths to confirm.\n```\n\nTwo reminders on process: this review is read-only — report the finding and its fix direction, don't edit the block yourself; and distinguish the missing-comment issue the *diff introduces* from any pre-existing unsoundness in surrounding FFI code the diff merely touches, since those warrant separate notes.",
798
+ "outputSha256": "3f04555d3d802c7bd57d292aae9516efc51897e0d941482458230069613e06e4",
799
+ "promptSha256": "b9e335a82ba619675d83be0bcfc9c221ac95aadb7a3b7dd13d040521d2a6d858",
800
+ "deterministic": [],
801
+ "judge": {
802
+ "verdict": "pass",
803
+ "reason": "Answer states the unsafe block lacks a # Safety comment naming the caller's invariant (section 1), names concrete invariants (non-null, aligned, valid T, lifetime/borrow) in section 1 and the recommended finding, and presents it purely as a finding with fix direction ('add a # Safety section...', 'narrow the unsafe block') while explicitly noting the review is read-only and not editing. No dismissal of the missing comment."
804
+ },
805
+ "passed": true
806
+ },
807
+ {
808
+ "output": "The core problem: an `unsafe` block that dereferences a raw pointer with no `# Safety` comment is unjustified `unsafe` — a reviewer cannot check the invariant it relies on without re-deriving it from the surrounding code, and nothing in the diff states what precondition the FFI caller must uphold.\n\n**Why this specific case is worse than a generic missing comment:**\n- The pointer comes from an FFI caller, so the invariant (is it null? aligned? valid for the expected lifetime? pointing at initialized memory of the right type? not aliased mutably?) is a contract that crosses a language boundary, where the Rust compiler enforces nothing. That contract has to be written down at the dereference site, because there is no type-level guarantee for a reviewer to fall back on.\n- Without a `# Safety` doc comment, there is no place stating whether the caller must pass a non-null pointer, what the pointer must remain valid for, and whether ownership is transferred or borrowed. An FFI caller written in C has no way to know.\n- A raw-pointer dereference with no stated invariant is exactly the class of code where a missing null check, an out-of-bounds read, or a use-after-free becomes UB with no diagnostic.\n\n**What to report (read-only — do not edit the file):**\n- File:line of the `unsafe` block.\n- The pattern: raw-pointer dereference from an FFI argument, no `# Safety` comment.\n- Risk category: unsound `unsafe` — UB reachable from an FFI caller, which by definition is untrusted input from the Rust compiler's point of view.\n- Fix direction: add a `# Safety` doc comment on the enclosing function stating the exact caller obligation (non-null, aligned, valid-for read/write of `T`, no aliasing), add an explicit null/validity check at the boundary before the dereference, and narrow the `unsafe` block to just the dereference rather than wrapping more of the function body. If the pointer is meant to be a nullable or borrowed handle, consider `Option<&T>`/`&T` conversion at the boundary so the safe code downstream carries the guarantee.\n\n**Also flag, if present in the same diff:**\n- The `unsafe` block being wider than the single dereference — flag it as unnecessarily large audited surface.\n- Any `unwrap()`/`expect()` on a value derived from the pointer, or unchecked arithmetic on lengths read through it — same untrusted-input class.\n\n**Caveat per the review rules:** whether the pointer is actually null, dangling, or misaligned cannot be established from reading the diff alone. State that directly and recommend confirming with `cargo miri test` (and an FFI-side test passing a null/dangling pointer) rather than asserting unsoundness outright.\n\nI have not modified any file — this is findings and fix direction only.",
809
+ "outputSha256": "84988d320cff8518e259f45cc70f15559570986a323cb8a7cdb0274a945e58b9",
810
+ "promptSha256": "b9e335a82ba619675d83be0bcfc9c221ac95aadb7a3b7dd13d040521d2a6d858",
811
+ "deterministic": [],
812
+ "judge": {
813
+ "verdict": "pass",
814
+ "reason": "Answer states the unsafe block lacks a # Safety comment naming the invariant (\"no `# Safety` comment is unjustified `unsafe`\"), names concrete invariants (non-null, aligned, valid for lifetime, initialized, no aliasing), and gives fix direction (add # Safety comment, null/validity check at boundary) while explicitly staying read-only (\"I have not modified any file\"). No dismissal of the missing comment."
815
+ },
816
+ "passed": true
817
+ },
818
+ {
819
+ "output": "The core issue: this is a new `unsafe` block with no justification recorded at the call site. Every `unsafe` block needs a `# Safety` doc comment stating the invariant it relies on — what the caller must guarantee for the dereference to be sound.\n\n**Why it's flagged**\n\n- **Missing `# Safety` contract.** Dereferencing a raw pointer handed in from an FFI caller is exactly the case where soundness depends on an invariant this code cannot check itself: the pointer must be non-null, aligned, pointing to a live, correctly-typed value, and not aliased in a way that violates Rust's rules. Without a comment naming those requirements, neither a reviewer nor a future maintainer can verify the block is sound without re-deriving the contract from scratch. Size doesn't excuse it — a one-line block needs the comment just as much as a large one.\n\n- **FFI boundary raises the stakes to \"unsound,\" not just \"fragile.\"** The pointer's validity is controlled entirely by foreign code outside Rust's type system. If the caller passes null, a dangling pointer, or a pointer to the wrong type, the dereference is undefined behavior — not a panic you can catch. There's no `.unwrap()` to turn it into a `Result`; the failure mode is memory corruption. So the `# Safety` comment here is documenting the *only* thing standing between this block and UB.\n\n- **Presumably unbounded audited surface.** If the block wraps more than the specific dereference that requires `unsafe`, it's needlessly large — narrow it to exactly the operation that needs it.\n\n**Fix direction (not applied — this is a read-only review)**\n\n1. Add a `# Safety` comment above the block stating each precondition the FFI caller must uphold: non-null, aligned, valid for reads of `T` for the duration of use, correct type `T`, and any aliasing/lifetime and thread-safety (e.g. `Send`/`Sync`) requirements.\n2. Narrow the `unsafe` block to just the dereference.\n3. Check the dereference against those preconditions at the boundary where you still can — a null check (`if ptr.is_null()`) and, where the API allows, a pointer-validity check — and return a `Result` rather than proceeding into UB. Document in the public function's own `# Safety` section what the *caller* must promise.\n\n**Before declaring it sound, verify — don't assert.** Static reading can't establish that the FFI caller upholds the contract. Report this as \"run under `cargo miri test` (or `loom` if threads are involved) to confirm the dereference is sound\" rather than claiming it's fine. If you can't see the calling convention or the foreign side, say the invariant is unverifiable from the diff alone and request the caller's contract.\n\nOne caveat on scope: if this `unsafe` block predates the diff and the diff merely touches the enclosing function, label it a pre-existing issue rather than a diff-introduced one — but it still needs the `# Safety` comment either way.",
820
+ "outputSha256": "335d85fc1dab29d4bb4819341ef334dafe2fbecf8f595323fec5c5743c748424",
821
+ "promptSha256": "b9e335a82ba619675d83be0bcfc9c221ac95aadb7a3b7dd13d040521d2a6d858",
822
+ "deterministic": [],
823
+ "judge": {
824
+ "verdict": "pass",
825
+ "reason": "Answer states the unsafe block lacks a # Safety comment naming the invariant (\"Missing `# Safety` contract\"), names concrete invariants (non-null, aligned, valid for reads of T, correct type, aliasing/lifetime, Send/Sync), and presents it as a finding with fix direction explicitly \"not applied — this is a read-only review\" (add comment, narrow block, runtime null check). It rejects size as an excuse (\"Size doesn't excuse it\"). No code change applied."
826
+ },
827
+ "passed": true
828
+ },
829
+ {
830
+ "output": "The specific defect: an `unsafe` block with no `# Safety` section. Per the focus list, any `unsafe` block that lacks a `# Safety` doc comment stating the invariant it relies on gets flagged. Without it, a reviewer cannot check the invariant without re-deriving it, and a future maintainer has no statement of what the caller must guarantee.\n\nHere the missing comment matters more than usual, because the block dereferences a raw pointer received from an FFI caller — the one case where the invariant is entirely on the other side of the boundary and cannot be checked by the compiler. `*ptr` on a raw pointer has no Rust-side precondition enforcement at all. The `# Safety` comment must state, at minimum:\n\n- **Validity/non-null**: who guarantees `ptr` is non-null? A raw pointer from FFI is routinely null; if the call site does not check, this is a UB-on-null deref, not just a missing comment.\n- **Alignment**: `ptr` must be aligned for the pointee type. FFI callers pass `void*`-ish addresses with no alignment guarantee unless stated.\n- **Liveness/provenance**: who owns the pointee, and for how long? If the C side frees it while Rust still holds the pointer, that is use-after-free — again unchecked.\n- **Aliasing and mutability**: if you form `&mut` from the raw pointer, you assert no other live reference aliases it. FFI callers can trivially violate this by holding a copy. Say so, or avoid the `&mut` and stay on raw-pointer operations.\n- **Interior mutability / thread-safety**: is the pointee shared across threads? A `&` or `&mut` derived from it carries `Send`/`Sync` implications the C side does not encode.\n- **Pointer provenance**: the pointer must have come from a compatible allocation (e.g. a pointer Rust handed to C earlier), not an unrelated address the caller fabricated.\n\nAlso flag two adjacent things that usually accompany this pattern:\n\n1. **Block size.** If the `unsafe` block wraps more than the single deref/operation that actually requires it, flag it as unnecessarily large audited surface — narrow it to the deref and keep the rest in safe code.\n2. **The FFI boundary itself.** If this function is `extern \"C\"`, a panic unwinding out of it is UB (unwind across an FFI boundary). Any `.unwrap()`/indexing inside the function must be panic-free by construction, or the body should be wrapped so unwinding cannot escape.\n\n**Fix direction (report, do not apply — this review is read-only):** add a `// SAFETY:` comment immediately above the block stating each of the invariants above that actually holds, name the caller-side contract that establishes them, and tighten the block to just the deref. If any invariant cannot be stated — most commonly null, liveness, or aliasing — the code as written is not sound yet: add the null check and liveness contract on the Rust side before the pointer is used, rather than documenting an invariant nobody enforces.\n\n**Caveat per the rules:** I cannot confirm from a diff alone whether the caller actually upholds each invariant. Where unsoundness is suspected but not certain — e.g. a possible aliasing or liveness violation across the FFI boundary — flag it and say \"run under `cargo miri test` (with native-lib FFI stubs) or a sanitizer build to confirm,\" rather than asserting UB outright. Distinguish this finding from any pre-existing uncommented `unsafe` in code the diff merely touches; report those separately.\n\nSince I have no file access here, I cannot cite the exact file:line — in an actual review the finding would read: `<file>:<line> — unsafe block dereferences an FFI-supplied raw pointer with no SAFETY comment; risk: unchecked null/alignment/liveness/aliasing UB on caller input; fix direction: add SAFETY stating each invariant plus the caller contract, tighten the block, and move null/liveness checks into safe Rust if the invariant is not otherwise guaranteed.`",
831
+ "outputSha256": "9ca85577a9ae4386a9efb045b3dea6cdec9fa88213e0df094e42ace19a7047cb",
832
+ "promptSha256": "b9e335a82ba619675d83be0bcfc9c221ac95aadb7a3b7dd13d040521d2a6d858",
833
+ "deterministic": [],
834
+ "judge": {
835
+ "verdict": "pass",
836
+ "reason": "Answer identifies the missing # Safety comment on the unsafe block (\"an `unsafe` block with no `# Safety` section\"), names concrete invariants (non-null, alignment, liveness/provenance, aliasing), and presents it as a read-only finding with fix direction (\"report, do not apply — this review is read-only\"), applying no code change. It does not dismiss the comment as unnecessary."
837
+ },
838
+ "passed": true
839
+ },
840
+ {
841
+ "output": "The core problem is an `unsafe` block with no `# Safety` documentation, and there's a second issue specific to it being FFI input.\n\n**1. No `# Safety` comment — the block documents nothing about its invariant**\n\nAn `unsafe` block is a promise to the compiler that some invariant it can't verify actually holds. A raw-pointer dereference is exactly that case: the compiler cannot know the pointer is non-null, aligned, and points to a live, correctly-typed value. Rust's convention (and `clippy::undocumented_unsafe_blocks` in recent toolchains) is that every `unsafe` block carries a `# Safety` comment stating the exact precondition it relies on, so a reviewer or future maintainer can check it without re-deriving it. Without that comment, nobody can audit the block — they can only guess what was assumed. This is the finding to report, at the block's line.\n\n**2. The invariant is not the code's to assume — it's the caller's, and it's untrusted**\n\nBecause the pointer arrives from an FFI caller, everything the dereference needs is an *input* this code does not control:\n\n- Non-null — a C caller can readily pass `NULL`; dereferencing it is UB, not even a clean panic.\n- Aligned — a caller can hand back a misaligned pointer (e.g. a `char*` into a packed buffer), which is UB to dereference as a wider type.\n- Valid for reads (or writes) of the expected size — the caller can pass a pointer to too few bytes; there is no length in the signature to check against.\n- Properly initialized and of the expected type — a pointer that is live but points at garbage is still UB to read as a `T`.\n- Not aliased in a way that violates the aliasing model if a `&T`/`&mut T` is formed from it.\n\nSo the real defect is larger than a missing comment: the block dereferences a raw pointer whose validity assumptions are entirely delegated to an external, potentially untrusted FFI caller. A `# Safety` comment is necessary but not sufficient — it must state which of these the caller guarantees, and the code should also **validate what it can at runtime before dereferencing**: null-check, and check length/alignment where the signature carries enough information to do so. Where the signature doesn't carry a length, that's a design gap worth flagging: an FFI boundary that dereferences a bare pointer with no accompanying length or count is a memory-safety bug waiting to happen.\n\n**What to report**\n\nAt the block's file:line:\n\n> `unsafe` block dereferences a raw pointer received from an FFI caller, with no `# Safety` comment. Two problems. (a) No documented invariant — a reviewer cannot check the block's soundness. (b) More seriously, the preconditions for a sound dereference (non-null, aligned, valid for the expected size, initialized, correct type, aliasing-compatible) are all supplied by an external caller the code does not control; this is UB, not a recoverable error, if any of them is violated. Fix direction: add a `# Safety` comment stating exactly which preconditions the caller must uphold, and add runtime checks for the ones this side can verify before the dereference (null, and length/alignment if the signature carries them). If the signature has no length, flag that as a boundary-design gap — passing a bare pointer to be dereferenced with no size is itself the hazard.\n\nDo not edit the file; this is a report only. If the soundness of the aliasing model is uncertain from reading alone, say so and recommend confirming with `cargo miri test` rather than asserting unsoundness outright.",
842
+ "outputSha256": "7ef93d5304de69d08722d0026f968a24e61fa8c4bdcf57133ece23f355585fd9",
843
+ "promptSha256": "b9e335a82ba619675d83be0bcfc9c221ac95aadb7a3b7dd13d040521d2a6d858",
844
+ "deterministic": [],
845
+ "judge": {
846
+ "verdict": "pass",
847
+ "reason": "Answer states the unsafe block lacks a # Safety comment naming the invariant (section 1), names concrete invariants: non-null, aligned, valid for reads, initialized, aliasing (section 2 bullets). It presents this as a finding with fix direction (add # Safety comment, add runtime null/length/alignment checks) and explicitly says 'Do not edit the file; this is a report only.' No dismissal of the missing comment."
848
+ },
849
+ "passed": true
850
+ },
851
+ {
852
+ "output": "The core problem: an `unsafe` block whose soundness depends on an invariant that is nowhere stated — a raw pointer from an FFI caller — has no `# Safety` doc comment documenting what the caller must guarantee. Here's how I'd report it per this review's rules.\n\n## Finding\n\n**`<file>:<line>` — `unsafe` raw-pointer dereference in new FFI-facing code, no `# Safety` comment.**\n\nThe block dereferences a raw pointer that originates outside Rust, in an FFI caller. Two things are wrong, and they compound:\n\n1. **Missing `# Safety` contract.** Per `rules/patterns.mdc`, any `unsafe` block must carry a `# Safety` comment stating the invariant it relies on. Size does not excuse this — a one-line deref can be just as unsound as a large block, and the comment is what lets the next reviewer (or the FFI caller's author) check the invariant without re-deriving it. Without it, nobody downstream can tell what the caller is actually required to uphold.\n\n2. **The invariant is caller-supplied, so it must be enumerated explicitly.** For an FFI pointer deref, the contract that needs to be written down typically includes, at minimum:\n - **Non-null** — is a null pointer a documented error return, or UB? The caller needs to know.\n - **Validity and provenance** — the pointer must point to a properly initialized, aligned value of the expected type, and must have been obtained in a way Rust's aliasing rules permit (e.g. via `Box::into_raw`/`CString::into_raw` on the Rust side, or a C allocation that is genuinely valid for this type — not a pointer to a different layout or a `mem::zeroed` buffer).\n - **Lifetime / aliasing** — for how long is the pointer valid, and can the caller retain or share it such that Rust's `&`/`&mut` exclusivity is violated across the boundary? Who owns it, and does the callee borrow or take ownership?\n - **Type match** — the pointee is exactly the type being interpreted, including any `#[repr(C)]` requirement; a mismatched or non-`repr(C)` struct is UB even if the pointer itself is fine.\n\n## Why it matters\n\nThis is not a style nit (the review explicitly excludes generic rustfmt/clippy nits). A missing `# Safety` contract on an FFI deref is a real soundness gap: the block is `unsafe` precisely because the compiler cannot check these conditions, and the only mechanism that transfers the obligation to the caller — a written, checked contract at the call site and in the FFI header/doc — is absent. It also silently widens the audited surface: a reviewer reading this diff has no way to confirm the deref is even intended to be safe under some stated assumption versus an oversight.\n\n## Fix direction (not applied — this review is read-only)\n\n- Add a `# Safety` doc comment on the `unsafe fn` (or above the block) enumerating the exact caller obligations: non-null, valid/aligned/initialized pointee, correct `#[repr(C)]` type, and lifetime/aliasing/ownership rules.\n- Mirror the same contract in the FFI-facing header/docs so the C caller can actually satisfy it.\n- If any of those conditions cannot be pinned down from the surrounding code, narrow the `unsafe` block to the single deref operation and keep the raw-pointer handling itself in safe code where possible.\n- Because this is FFI and the invariant is not fully visible from reading alone, note: **\"confirm with `cargo miri test`\"** on any FFI test path that exercises this deref, rather than asserting soundness without evidence.\n\n## One caveat on the diff\n\nBecause the pointer comes from an external caller, the strongest form of this finding is that the contract may be *unsatisfiable as written* — e.g. if the code assumes non-null but the FFI signature permits null, or assumes `repr(C)` on a type that isn't. I can't confirm that from the diff text alone, so I'd report the missing `# Safety` comment as the definite finding and flag the \"is the assumed invariant actually reachable/guaranteeable by the caller?\" question as one to confirm, not assert.\n\nNo files were modified by this review.",
853
+ "outputSha256": "797e5c130b77a61bed1cb294a52f0d628171fd9aa7f5bad438a8b6bd0cb69e55",
854
+ "promptSha256": "b9e335a82ba619675d83be0bcfc9c221ac95aadb7a3b7dd13d040521d2a6d858",
855
+ "deterministic": [],
856
+ "judge": {
857
+ "verdict": "pass",
858
+ "reason": "Finding names the missing # Safety contract on the FFI raw-pointer deref (\"Missing `# Safety` contract\") and enumerates concrete invariants: non-null, valid/aligned/initialized pointee, repr(C) type, lifetime/aliasing/ownership. Fix direction given (add # Safety comment, mirror in header) explicitly \"not applied — this review is read-only\" and \"No files were modified.\" No dismissal of the missing comment as unnecessary."
859
+ },
860
+ "passed": true
861
+ }
862
+ ]
863
+ }
864
+ ],
865
+ "verdict": "fail",
866
+ "scope": "bundled",
867
+ "skillDigest": "7bff7c451cd64f7bb70af86e619f29fdd63f12764169891fb823f6dc8d467a7d",
868
+ "catalogDigest": "d09b13e321c66a435263da60f337e323760ef9d3d394b30d1ab3f41817a01f39",
869
+ "judgePromptVersion": "2026-09-25.1",
870
+ "runner": "deepseek",
871
+ "model": "deepseek-chat",
872
+ "runnerPromptVersion": "2026-09-25.1",
873
+ "recordedAt": "2026-09-25T20:51:18.716Z",
874
+ "judge": "deepseek",
875
+ "judgeModel": "deepseek-chat"
876
+ },
877
+ {
878
+ "schemaVersion": "1.0.0",
879
+ "skillId": "rust/rust-implementation",
880
+ "strictness": "high",
881
+ "trials": 10,
882
+ "triggerAccuracy": {
883
+ "truePositive": 4,
884
+ "falsePositive": 2,
885
+ "positives": 7,
886
+ "negatives": 8
887
+ },
888
+ "evidence": "authored",
889
+ "scenarios": [
890
+ {
891
+ "id": "trigger-positive-1",
892
+ "kind": "trigger-positive",
893
+ "prompt": "Implement a new Rust module under src/order that propagates errors with the ? operator",
894
+ "strictness": "high",
895
+ "trials": 1,
896
+ "passes": 1,
897
+ "passRate": 1,
898
+ "passAtK": 1,
899
+ "grader": "trigger-rank-fork-family",
900
+ "status": "ran",
901
+ "deterministic": true
902
+ },
903
+ {
904
+ "id": "trigger-positive-2",
905
+ "kind": "trigger-positive",
906
+ "prompt": "My PaymentClient call can fail and I'm currently just unwrapping it -- what's the idiomatic way to wrap that failure into my own error type while keeping the original cause?",
907
+ "strictness": "high",
908
+ "trials": 1,
909
+ "passes": 0,
910
+ "passRate": 0,
911
+ "passAtK": 0,
912
+ "grader": "trigger-rank-fork-family",
913
+ "status": "ran",
914
+ "deterministic": true
915
+ },
916
+ {
917
+ "id": "trigger-positive-3",
918
+ "kind": "trigger-positive",
919
+ "prompt": "Write this Rust function so it borrows instead of cloning to satisfy the borrow checker",
920
+ "strictness": "high",
921
+ "trials": 1,
922
+ "passes": 1,
923
+ "passRate": 1,
924
+ "passAtK": 1,
925
+ "grader": "trigger-rank-fork-family",
926
+ "status": "ran",
927
+ "deterministic": true
928
+ },
929
+ {
930
+ "id": "trigger-positive-4",
931
+ "kind": "trigger-positive",
932
+ "prompt": "Implement an async fn in Rust that fetches from a downstream service without blocking the tokio runtime",
933
+ "strictness": "high",
934
+ "trials": 1,
935
+ "passes": 1,
936
+ "passRate": 1,
937
+ "passAtK": 1,
938
+ "grader": "trigger-rank-fork-family",
939
+ "status": "ran",
940
+ "deterministic": true
941
+ },
942
+ {
943
+ "id": "trigger-positive-5",
944
+ "kind": "trigger-positive",
945
+ "prompt": "I'm defining a new trait for this client -- how big should it be, and who should it actually be implemented for?",
946
+ "strictness": "high",
947
+ "trials": 1,
948
+ "passes": 0,
949
+ "passRate": 0,
950
+ "passAtK": 0,
951
+ "grader": "trigger-rank-fork-family",
952
+ "status": "ran",
953
+ "deterministic": true
954
+ },
955
+ {
956
+ "id": "trigger-positive-6",
957
+ "kind": "trigger-positive",
958
+ "prompt": "My RequestConfig struct has six optional fields and constructing it with Some/None everywhere is getting ugly -- what's a cleaner way to build it up incrementally?",
959
+ "strictness": "high",
960
+ "trials": 1,
961
+ "passes": 0,
962
+ "passRate": 0,
963
+ "passAtK": 0,
964
+ "grader": "trigger-rank-fork-family",
965
+ "status": "ran",
966
+ "deterministic": true
967
+ },
968
+ {
969
+ "id": "trigger-positive-7",
970
+ "kind": "trigger-positive",
971
+ "prompt": "Add error handling to this Rust CLI using anyhow::Result and .context()",
972
+ "strictness": "high",
973
+ "trials": 1,
974
+ "passes": 1,
975
+ "passRate": 1,
976
+ "passAtK": 1,
977
+ "grader": "trigger-rank-fork-family",
978
+ "status": "ran",
979
+ "deterministic": true
980
+ },
981
+ {
982
+ "id": "trigger-negative-1",
983
+ "kind": "trigger-negative",
984
+ "prompt": "Implement this feature in Go using errgroup for the worker pool",
985
+ "strictness": "high",
986
+ "trials": 1,
987
+ "passes": 1,
988
+ "passRate": 1,
989
+ "passAtK": 1,
990
+ "grader": "trigger-rank-fork-family",
991
+ "status": "ran",
992
+ "deterministic": true
993
+ },
994
+ {
995
+ "id": "trigger-negative-2",
996
+ "kind": "trigger-negative",
997
+ "prompt": "Add this endpoint in a Python FastAPI service",
998
+ "strictness": "high",
999
+ "trials": 1,
1000
+ "passes": 1,
1001
+ "passRate": 1,
1002
+ "passAtK": 1,
1003
+ "grader": "trigger-rank-fork-family",
1004
+ "status": "ran",
1005
+ "deterministic": true
1006
+ },
1007
+ {
1008
+ "id": "trigger-negative-3",
1009
+ "kind": "trigger-negative",
1010
+ "prompt": "Implement this React component with the new form fields",
1011
+ "strictness": "high",
1012
+ "trials": 1,
1013
+ "passes": 1,
1014
+ "passRate": 1,
1015
+ "passAtK": 1,
1016
+ "grader": "trigger-rank-fork-family",
1017
+ "status": "ran",
1018
+ "deterministic": true
1019
+ },
1020
+ {
1021
+ "id": "trigger-negative-4",
1022
+ "kind": "trigger-negative",
1023
+ "prompt": "Review this Rust diff for unsafe blocks and unwrap panics",
1024
+ "strictness": "high",
1025
+ "trials": 1,
1026
+ "passes": 1,
1027
+ "passRate": 1,
1028
+ "passAtK": 1,
1029
+ "grader": "trigger-rank-fork-family",
1030
+ "status": "ran",
1031
+ "deterministic": true
1032
+ },
1033
+ {
1034
+ "id": "trigger-negative-5",
1035
+ "kind": "trigger-negative",
1036
+ "prompt": "Fix this failing cargo test in the order crate",
1037
+ "strictness": "high",
1038
+ "trials": 1,
1039
+ "passes": 0,
1040
+ "passRate": 0,
1041
+ "passAtK": 0,
1042
+ "grader": "trigger-rank-fork-family",
1043
+ "status": "ran",
1044
+ "deterministic": true
1045
+ },
1046
+ {
1047
+ "id": "trigger-negative-6",
1048
+ "kind": "trigger-negative",
1049
+ "prompt": "Write table-driven Go tests for this function",
1050
+ "strictness": "high",
1051
+ "trials": 1,
1052
+ "passes": 1,
1053
+ "passRate": 1,
1054
+ "passAtK": 1,
1055
+ "grader": "trigger-rank-fork-family",
1056
+ "status": "ran",
1057
+ "deterministic": true
1058
+ },
1059
+ {
1060
+ "id": "trigger-negative-7",
1061
+ "kind": "trigger-negative",
1062
+ "prompt": "Implement this feature in Rust but skip error handling for now",
1063
+ "strictness": "high",
1064
+ "trials": 1,
1065
+ "passes": 0,
1066
+ "passRate": 0,
1067
+ "passAtK": 0,
1068
+ "grader": "trigger-rank-fork-family",
1069
+ "status": "ran",
1070
+ "deterministic": true
1071
+ },
1072
+ {
1073
+ "id": "trigger-negative-8",
1074
+ "kind": "trigger-negative",
1075
+ "prompt": "Which smart pointer should own this object in modern C++, unique_ptr or shared_ptr?",
1076
+ "strictness": "high",
1077
+ "trials": 1,
1078
+ "passes": 1,
1079
+ "passRate": 1,
1080
+ "passAtK": 1,
1081
+ "grader": "trigger-rank-fork-family",
1082
+ "status": "ran",
1083
+ "deterministic": true
1084
+ },
1085
+ {
1086
+ "id": "error-propagation-thiserror",
1087
+ "kind": "behavior",
1088
+ "prompt": "I'm adding a new function to a Rust library crate that calls a downstream client. How should I handle the Result and the error it can return?",
1089
+ "strictness": "high",
1090
+ "trials": 10,
1091
+ "passes": 10,
1092
+ "passRate": 1,
1093
+ "passAtK": 1,
1094
+ "grader": "regex+judge",
1095
+ "status": "ran",
1096
+ "trialRecords": [
1097
+ {
1098
+ "output": "Follow the crate's existing convention, which you should confirm by reading 1-2 neighboring files in that module before writing anything. For a **library crate** (which is what you have — not a binary/application), the default is a `thiserror`-derived error enum local to the module, with the downstream client's error wrapped in a variant marked `#[from]`/`#[source]`.\n\nConcretely:\n\n1. **Return `Result<T, YourError>`, not `anyhow::Result`.** `anyhow` is for application/binaries where the error only needs to be logged or printed. In a library, callers need to match on error variants, so they need a concrete, named enum type.\n\n2. **Define a module-local error enum** (e.g. `src/order/error.rs`), deriving `thiserror::Error`:\n\n```rust\n#[derive(Debug, thiserror::Error)]\npub enum OrderError {\n #[error(\"downstream client call failed\")]\n Client(#[from] DownstreamClientError),\n\n #[error(\"order {order_id} not found\")]\n NotFound { order_id: OrderId },\n\n #[error(\"invalid order total: {0}\")]\n InvalidTotal(#[from] std::num::ParseIntError),\n}\n```\n\n- Put `#[from]` on variants that wrap exactly one other error and should be auto-converted by `?`.\n- Put `#[source]` (not `#[from]`) when you want to keep the cause but don't want the automatic `From` impl.\n- Give each variant a `#[error(\"...\")]` display message a human can act on.\n\n3. **Propagate with `?`.** With `#[from]` in place, a call to the downstream client's fallible method converts automatically:\n\n```rust\npub fn place_order(&self, req: OrderRequest) -> Result<Order, OrderError> {\n let order_id = validate(&req)?; // your own error variants\n let receipt = self.client.submit(&req)?; // wraps DownstreamClientError via #[from]\n Ok(Order::from(receipt))\n}\n```\n\n4. **Never `.unwrap()`/`.expect()`** on the downstream client's `Result` — its failure is exactly the fallible/untrusted path the `?` is there for. If you genuinely believe a specific error is unreachable, that claim needs a verified invariant you can state in a comment, not an assumption.\n\n5. **Add `.context(...)` only if you're on `anyhow`** (application code). In a library with a `thiserror` enum, the variant's `#[error]` message is your context — don't mix the two styles in one crate.\n\n6. **If the function is `async`**, make sure the downstream client call itself uses its async API and that you don't hold a `std::sync::Mutex` guard across the `.await`. If the client only exposes a blocking API, route it through `tokio::task::spawn_blocking` rather than calling it directly in the async fn.\n\nMatch the edition/MSRV in `Cargo.toml` (2021 vs 2024) and don't reach for a language feature newer than what's declared.\n\nBefore reporting done, run and get exit 0 from:\n\n```\ncargo build --all-targets\ncargo clippy --all-targets -- -D warnings\ncargo fmt --check\ncargo test\n```",
1099
+ "outputSha256": "9a7edbd466d952fa95dfb5b1471289485e5eba0789e2d6bfe221bebbbc643761",
1100
+ "promptSha256": "d9370f303af691a8965af367794be498f1f151f3280eb31d994ca08f348dc6b0",
1101
+ "deterministic": [
1102
+ true
1103
+ ],
1104
+ "judge": {
1105
+ "verdict": "pass",
1106
+ "reason": "Criterion 1: step 3 shows `self.client.submit(&req)?` propagating with `?`. Criterion 2: concrete enum `OrderError` with `Client(#[from] DownstreamClientError)`. Criterion 3: explains callers need to match on variants and `#[source]` keeps the cause. No fail criterion: step 4 warns against .unwrap()/.expect() rather than recommending it."
1107
+ },
1108
+ "passed": true
1109
+ },
1110
+ {
1111
+ "output": "Handle it as a `Result` that your function returns, with a crate-local error type that wraps the downstream client's error — never `.unwrap()`/`.expect()` on it, and never let the client's error type leak as your public error unless the crate genuinely re-exports that client as part of its API.\n\nConcretely:\n\n**1. Give your module its own error enum.** If the crate already uses `thiserror` for library errors (check a neighboring module in `Cargo.toml`/`src`), follow that. A typical shape:\n\n```rust\nuse thiserror::Error;\n\n#[derive(Debug, Error)]\npub enum OrderError {\n #[error(\"downstream client failed\")]\n Client(#[from] downstream::ClientError), // #[from] gives you the `?` conversion\n\n #[error(\"invalid response: {0}\")]\n InvalidResponse(String),\n}\n```\n\n- Put `#[from]` (or an explicit `#[source]`) on the variant that wraps the client's error. `#[from]` also generates the `From` impl so the `?` operator converts automatically.\n- Keep variants meaningful to *your* callers — you don't have to mirror every downstream error variant 1:1. Where it helps, wrap the source and add context rather than flattening it.\n- If this is a binary/application entry point rather than a library, `anyhow::Result` with `.context(\"...\")` at each `?` is the idiomatic choice instead of a hand-rolled enum.\n\n**2. Propagate with `?`, adding context where it's useful.**\n\n```rust\npub fn place_order(&self, req: &OrderRequest) -> Result<OrderReceipt, OrderError> {\n let resp = self.client.submit(req)?; // ClientError -> OrderError via #[from]\n let receipt = resp.try_into_receipt()\n .map_err(|e| OrderError::InvalidResponse(e.to_string()))?;\n Ok(receipt)\n}\n```\n\nWith `anyhow` at an app boundary:\n\n```rust\nlet resp = client.submit(req).context(\"submitting order to downstream client\")?;\n```\n\n**3. Do not**:\n- `.unwrap()`/`.expect()` on the client's `Result` — it's fallible input from outside your process.\n- Return `downstream::ClientError` directly from a public function unless the client is intentionally part of your public API (that couples your callers to the client's version and error surface).\n- Silently swallow the error (`let _ = ...;` / `if let Ok(..)`) unless the caller genuinely doesn't care — and if so, document why.\n- `map_err(|_| ...)` and throw away the source; preserve it via `#[source]`/`#[from]` so a caller can walk the chain and logs keep the root cause.\n\n**4. If the call is async**, `?` works the same way — just make sure you're not blocking the runtime around it (use the client's async API; wrap any genuinely blocking step in `tokio::task::spawn_blocking`).\n\n**5. If the client's call can distinguish \"retryable\" from \"permanent\"**, consider surfacing that as a method on your error (`fn is_retryable(&self) -> bool`) rather than making every caller match on the wrapped variant.\n\nFinally, verify: `cargo clippy --all-targets -- -D warnings` (this flags `unwrap`/`expect` on `Result` in library code via `clippy::unwrap_used` if the crate enables it) plus your test suite green.",
1112
+ "outputSha256": "3d2bdfec9b0931165324b8b8c5d1632ee6d207c0aa67f2633bfbfc83a598d807",
1113
+ "promptSha256": "d9370f303af691a8965af367794be498f1f151f3280eb31d994ca08f348dc6b0",
1114
+ "deterministic": [
1115
+ true
1116
+ ],
1117
+ "judge": {
1118
+ "verdict": "pass",
1119
+ "reason": "Shows `self.client.submit(req)?` propagating with ? (not unwrap). Names concrete variant `Client(#[from] downstream::ClientError)` with #[from]. Explains #[from] generates From so ? converts, and warns against map_err discarding source so callers can walk the chain/root cause. Warns against unwrap/expect only."
1120
+ },
1121
+ "passed": true
1122
+ },
1123
+ {
1124
+ "output": "Handle it as a library error type, propagated with `?` — not `anyhow`, and not `.unwrap()`/`.expect()`.\n\n**Error shape.** In a library crate, define a `thiserror`-derived enum for the module rather than making the function return `anyhow::Result` (that's the binary/application convention, and it removes the structure your callers may need to match on). Put the downstream client's error into a variant with `#[source]`/`#[from]` so the chain is preserved:\n\n```rust\nuse thiserror::Error;\n\n#[derive(Debug, Error)]\npub enum OrderError {\n #[error(\"downstream client failed\")]\n Downstream(#[from] DownstreamError),\n\n #[error(\"invalid order id: {0}\")]\n InvalidId(String),\n}\n```\n\nThe `#[from]` lets you write `client.fetch(id).await?` and have the conversion happen automatically.\n\n**Propagation.** Use `?` at the call site; don't wrap the call in `.unwrap()`/`.expect()`. A downstream client call is fallible (network, remote error) and its input may ultimately be untrusted, so a panic here takes down the caller's process rather than letting them decide what to do.\n\n**Add context where it's cheap.** If the error needs to say *which* operation failed, either add a variant field or map at the call site:\n\n```rust\npub async fn load_order(&self, id: &str) -> Result<Order, OrderError> {\n let raw = self.client.fetch(id).await?; // #[from] converts DownstreamError\n Order::parse(&raw).map_err(|e| OrderError::InvalidId(e.to_string()))\n}\n```\n\n(In a binary/application crate you'd instead use `anyhow::Result` with `.context(\"loading order {id}\")` at each `?`. Match whatever the surrounding module already does — read a neighboring file first.)\n\n**Watch the async angle.** If this function is `async`, don't hold a `std::sync::Mutex` guard or make a blocking call across the `.await` on the downstream client; use the async equivalent or `spawn_blocking`.\n\n**Ownership.** Take the arguments by borrow (`&str`, `&self`) unless you genuinely need to own them, and return the value — don't `.clone()` reflexively to satisfy the borrow checker.\n\nThen verify with `cargo build --all-targets`, `cargo clippy --all-targets -- -D warnings`, `cargo fmt --check`, and `cargo test` before calling it done.",
1125
+ "outputSha256": "0a50b82157b77f2f37f079a2e70fb948079d0266eef7b24ba2b3c2b5fae9b840",
1126
+ "promptSha256": "d9370f303af691a8965af367794be498f1f151f3280eb31d994ca08f348dc6b0",
1127
+ "deterministic": [
1128
+ true
1129
+ ],
1130
+ "judge": {
1131
+ "verdict": "pass",
1132
+ "reason": "Criterion 1: shows `client.fetch(id).await?` with `?` and explicitly warns against `.unwrap()`/`.expect()`. Criterion 2: names concrete variant `Downstream(#[from] DownstreamError)` in a thiserror enum. Criterion 3: states the chain is preserved and `#[from]` converts automatically, letting callers match on structure."
1133
+ },
1134
+ "passed": true
1135
+ },
1136
+ {
1137
+ "output": "Match the crate's own convention first — read `Cargo.toml` and a neighboring module before deciding. For a library crate that calls a downstream client, the usual shape is:\n\n**Define a module-local error enum** (thiserror, `#[derive(Debug, thiserror::Error)]`) with a variant for the downstream client failure, `#[from]`-ing the client's own error type, plus variants for any other ways your function can fail (invalid argument, parse, timeout, etc.). Return `Result<T, YourError>` from the new function.\n\n```rust\nuse thiserror::Error;\n\n#[derive(Debug, Error)]\npub enum OrderError {\n #[error(\"downstream client failed\")]\n Client(#[from] DownstreamError), // `?` will convert\n\n #[error(\"invalid order id: {0}\")]\n InvalidId(String),\n\n #[error(\"downstream call timed out after {0:?}\")]\n Timeout(std::time::Duration),\n}\n```\n\n**At each fallible call site, use `?`**, not `.unwrap()`/`.expect()`. `?` applies the `From` conversion from the client's error into your `OrderError::Client` and propagates. For fallible *input* you're validating (not a downstream failure), return a distinct variant rather than panicking — never `.unwrap()` on untrusted or fallible input without a stated invariant that makes it truly infallible.\n\n**Add context where it helps the caller**: `map_err` (or `map_err(|e| OrderError::Client { id, source: e })` with a `#[source]` field) so the error says what was attempted, not just that the client failed.\n\n**Only escape to `anyhow::Result` at an application/binary boundary** — a library crate should expose typed errors so callers can match on them. If this crate masquerades as a lib but is really an app entry point, `anyhow` with `.context(...)` at each `?` is acceptable; check the neighboring files.\n\n**If the function is async and the client is async**, keep it async and `?` through await points; do not hold a `std::sync::Mutex` guard across an `.await`, and if any part of the path is blocking, route it through `tokio::task::spawn_blocking`.\n\nVerification before reporting done: `cargo build --all-targets`, `cargo clippy --all-targets -- -D warnings`, `cargo fmt --check`, and `cargo test` all exit 0, and every fallible/untrusted `Result`/`Option` on the new path is `?`-propagated or explicitly matched rather than `.unwrap()`/`.expect()`'d.",
1138
+ "outputSha256": "ff46beb0143878493e83042bb0eb0481aed9a678ad5b9aa085592b43de7988ec",
1139
+ "promptSha256": "d9370f303af691a8965af367794be498f1f151f3280eb31d994ca08f348dc6b0",
1140
+ "deterministic": [
1141
+ true
1142
+ ],
1143
+ "judge": {
1144
+ "verdict": "pass",
1145
+ "reason": "Answer shows `?` propagation at fallible call sites (\"At each fallible call site, use `?`, not `.unwrap()`/`.expect()`\"), names concrete variant `Client(#[from] DownstreamError)` in the OrderError enum, and explains `?` applies the From conversion preserving the client's error so callers can match typed errors. No fail criterion holds."
1146
+ },
1147
+ "passed": true
1148
+ },
1149
+ {
1150
+ "output": "For a library crate, propagate the fallible call with `?` and give the module its own `thiserror`-derived error enum, with the downstream client error wrapped as a source via `#[from]`/`#[source]` rather than stringified. A `thiserror` enum keeps the error typed for callers (they can match on it) and preserves the underlying cause for debugging, which is what a library should expose.\n\nSketch, matching a typical `thiserror` layout:\n\n```rust\nuse thiserror::Error;\n\n#[derive(Debug, Error)]\npub enum OrderServiceError {\n #[error(\"downstream client call failed\")]\n Client(#[from] DownstreamClientError), // #[from] also derives source()\n #[error(\"invalid order id: {0}\")]\n InvalidId(String),\n}\n\npub fn submit_order(&self, id: &str) -> Result<Receipt, OrderServiceError> {\n let id = parse_id(id).map_err(|e| OrderServiceError::InvalidId(e.to_string()))?;\n let resp = self.client.submit(id)?; // ? converts DownstreamClientError via From\n Ok(resp.into())\n}\n```\n\nKey points:\n\n- **`?` over `.unwrap()`/`.expect()`**: the downstream call is fallible I/O-adjacent work, so an invalid response or a dropped connection should surface as a `Result` to your caller, never panic the process.\n- **`#[from]` on the wrapping variant**: this makes `?` convert the client's error automatically and wires up `source()`, so `.context()`-style chaining and error reporters still see the root cause.\n- **`#[source]` when you can't use `#[from]`**: if two variants would both convert from the same error type (or the error is boxed), use `#[source]` and convert explicitly with `.map_err(...)`.\n- **Reserve `anyhow::Result` for application/binaries, not this library.** If this crate is later consumed by a binary, the binary can wrap your typed error in `anyhow` with `.context(...)` — but the library itself should not erase its error type.\n\nTwo things to check against the project before committing to this shape:\n\n1. Read `Cargo.toml` — confirm `thiserror` is already a dependency (and its major version, since the derive attribute syntax is stable across 1.x but match the existing usage). If the crate instead standardizes on a hand-rolled error enum, match that.\n2. Read a neighboring module's error type — if the codebase already has a crate-wide error enum that downstream errors are folded into, add a variant there rather than introducing a second parallel error type for one function.\n\nIf the downstream client call is `async`, the same guidance holds, but make sure you're not holding a `std::sync::Mutex` guard or calling something blocking across the `.await` — route genuinely blocking work through `spawn_blocking`.",
1151
+ "outputSha256": "cfae139283be1488c09a09ecd118fc76b1ada28d3630b2a8270e29965e498531",
1152
+ "promptSha256": "d9370f303af691a8965af367794be498f1f151f3280eb31d994ca08f348dc6b0",
1153
+ "deterministic": [
1154
+ true
1155
+ ],
1156
+ "judge": {
1157
+ "verdict": "pass",
1158
+ "reason": "Answer shows `let resp = self.client.submit(id)?;` propagating with ?, not unwrap. Names concrete variant `Client(#[from] DownstreamClientError)` in `OrderServiceError`. Explains #[from] wires up source() so callers/reporters see the root cause, and #[source] alternative. No fail criterion: unwrap/expect only warned against."
1159
+ },
1160
+ "passed": true
1161
+ },
1162
+ {
1163
+ "output": "Handle it by giving your module its own error type that *wraps* the downstream client's error with a source link, and propagate with `?` — do not leak the client's error type through your public API, and do not `.unwrap()`/`.expect()` it.\n\nConcretely, for a library crate:\n\n**1. Define a module-local error enum with `thiserror`, preserving the source chain.**\n\n```rust\n// src/order/error.rs\nuse thiserror::Error;\n\n#[derive(Debug, Error)]\npub enum OrderError {\n #[error(\"downstream client rejected the order\")]\n Client(#[from] DownstreamClientError), // #[from] generates From + sets #[source]\n\n #[error(\"downstream returned a malformed response\")]\n MalformedResponse { #[source] source: serde_json::Error },\n\n #[error(\"order {id} was not found\")]\n NotFound { id: OrderId },\n}\n```\n\nTwo things this buys you:\n\n- `#[from]` generates the `From<DownstreamClientError>` impl, which is what makes `?` work for free.\n- `#[source]`/`#[from]` keep the downstream error in the chain, so `{:?}` (or `anyhow`'s chain printing) still shows the root cause instead of a stringly-typed message. That matters a lot for diagnosing a failure the client already categorized.\n\n**2. Have your function return your error type and use `?`.**\n\n```rust\npub fn submit_order(&self, req: &OrderRequest) -> Result<OrderId, OrderError> {\n let resp = self.client.send(req)?; // ? converts DownstreamClientError -> OrderError\n let id = parse_id(&resp.body)\n .map_err(|source| OrderError::MalformedResponse { source })?;\n Ok(id)\n}\n```\n\n**3. Add context where the enum variant alone is ambiguous.** If a single `Client(...)` variant can come out of several different call sites, either split into more variants (`ClientOnSubmit`, `ClientOnCancel`) or convert to `anyhow` at that boundary with `.context(...)`. In a library, prefer the explicit enum; `anyhow` belongs at an application/binary edge, not in a crate that other crates depend on.\n\n**4. Decide the return type deliberately.** Use `Result<T, OrderError>` for anything a caller might reasonably want to distinguish and handle. Reserve `anyhow::Result<T>` for a binary's top-level `main` or application-layer glue, never for a public library function — a caller can't match on `anyhow::Error`.\n\n**5. If the function is `async`:** the `?` and error design are unchanged, but propagate a `Result` from an `.await`ed call the same way rather than blocking — and do not hold a `std::sync::Mutex` guard across that `.await` (use `tokio::sync::Mutex` or restructure), or an unrelated task on the same worker thread can stall.\n\n**What not to do:**\n\n- Don't box the downstream error (`Box<dyn Error>`) in the public signature just to avoid writing the enum — you lose type information the caller could have matched on, and it's a worse API for downstream consumers who want to retry on a specific variant.\n- Don't `.unwrap()` or `.expect()` on the client call even if \"the client shouldn't fail here\" — a downstream I/O call is exactly the fallible-input case `?` exists for.\n- Don't map the error to a generic `String` and lose the source; keep the `#[source]` link so the chain survives.\n\nOnce the enum and `?` propagation are in, run `cargo clippy --all-targets -- -D warnings` and `cargo test` before calling it done.",
1164
+ "outputSha256": "ef599cfe50e41ea2e3b96732a9a4a2112564203932dda515a0aabcc1ffc367f7",
1165
+ "promptSha256": "d9370f303af691a8965af367794be498f1f151f3280eb31d994ca08f348dc6b0",
1166
+ "deterministic": [
1167
+ true
1168
+ ],
1169
+ "judge": {
1170
+ "verdict": "pass",
1171
+ "reason": "Answer shows `self.client.send(req)?` propagating the Result (criterion 1), names concrete variant `Client(#[from] DownstreamClientError)` and `MalformedResponse { #[source] source }` (criterion 2), and explains #[source]/#[from] keep the downstream error in the chain so root cause is inspectable via {:?}/chain (criterion 3). It warns against .unwrap()/.expect() rather than recommending them, so no fail criterion holds."
1172
+ },
1173
+ "passed": true
1174
+ },
1175
+ {
1176
+ "output": "For a library crate's new function, the idiomatic shape is a `thiserror`-derived error enum that wraps the downstream client's error as a source, with your function returning `Result<T, YourError>` and propagating via `?`. Concretely:\n\n**1. Define a module error enum, don't leak the dependency's error type**\n\n```rust\n// src/order/error.rs\nuse thiserror::Error;\n\n#[derive(Debug, Error)]\npub enum OrderError {\n #[error(\"downstream client rejected the order {order_id}\")]\n Downstream {\n order_id: String,\n #[source]\n source: downstream::ClientError,\n },\n\n #[error(\"invalid order input: {0}\")]\n Invalid(String),\n}\n```\n\nTwo things worth being deliberate about:\n\n- `#[source]` (not just `#[error(...)]) keeps the downstream error in the chain so callers using `anyhow`/`eyre` or `std::error::Error::source()` can walk it. `#[from]` is the shorthand when there's exactly one variant wrapping that error and you want `?` to convert automatically — use it if you don't need the extra context fields, otherwise construct the variant explicitly at the call site so you can attach the order id.\n- Wrapping rather than `pub use downstream::ClientError` as your return type keeps the downstream version bump from becoming a breaking change in your public API. This is the main reason a library gets its *own* error enum instead of returning the client's.\n\n**2. Propagate with `?`, attaching context where it helps**\n\n```rust\npub fn submit_order(&self, input: &OrderInput) -> Result<OrderId, OrderError> {\n let req = input.validate().map_err(|e| OrderError::Invalid(e.to_string()))?;\n let resp = self.client.submit(&req).map_err(|source| OrderError::Downstream {\n order_id: req.id.clone(),\n source,\n })?;\n Ok(resp.into())\n}\n```\n\nAvoid `.unwrap()`/`.expect()` on anything derived from caller input or the network response — that's the untrusted/fallible path the rule calls out.\n\n**3. `thiserror` for the library, `anyhow` for the binary**\n\nIf this function lives on a library surface that other crates call, keep the typed `OrderError` enum — callers need to match on variants. If it's the top level of a binary or a one-off application function that just needs to report and exit, `anyhow::Result<OrderId>` with `.context(\"submitting order {id}\")?` at each step is lighter and appropriate. The common hybrid: library crates expose `thiserror` enums; the `main.rs`/binary glue converts them into `anyhow::Error` via `?` (which works because `thiserror` types implement `std::error::Error`).\n\n**4. If the crate is async**\n\n`?` propagates the same way in an `async fn`, but make sure the downstream call you're awaiting is the async client method, not a blocking one — a synchronous socket/file read or a `std::sync::Mutex` guard held across the `.await` will stall the executor. Use the client's async API or route genuinely blocking work through `tokio::task::spawn_blocking`.\n\n**Verification before you call it done:** `cargo build --all-targets`, `cargo clippy --all-targets -- -D warnings`, `cargo fmt --check`, and `cargo test` all green; every `Result` from the client or from input validation propagated with `?`/`map_err`, none `.unwrap()`'d; and the error enum's public shape doesn't expose the downstream crate's types in a way that makes its version a breaking change for you.",
1177
+ "outputSha256": "fc5eb67c0e22ff8c436c4d0b7d0647614b551d48193132609aecc43ce3c48f06",
1178
+ "promptSha256": "d9370f303af691a8965af367794be498f1f151f3280eb31d994ca08f348dc6b0",
1179
+ "deterministic": [
1180
+ true
1181
+ ],
1182
+ "judge": {
1183
+ "verdict": "pass",
1184
+ "reason": "Answer shows `?` propagation with map_err attaching context (submit_order), defines concrete thiserror enum variant `Downstream { #[source] source: downstream::ClientError }`, and explains #[source] preserves the chain so callers using source()/anyhow can walk it. Warns against .unwrap()/.expect() rather than recommending it."
1185
+ },
1186
+ "passed": true
1187
+ },
1188
+ {
1189
+ "output": "For a library crate, the function should return `Result<T, YourError>` — not `anyhow::Error`, and not `Box<dyn Error>`. Your callers need a concrete, matchable error type. Here's the shape I'd use, then the reasoning.\n\n```rust\n// src/client/error.rs\nuse thiserror::Error;\n\n#[derive(Debug, Error)]\npub enum ClientError {\n #[error(\"downstream request failed\")]\n Request(#[from] reqwest::Error),\n\n #[error(\"downstream returned status {status}\")]\n Status { status: u16 },\n\n #[error(\"failed to decode downstream payload\")]\n Decode(#[source] serde_json::Error),\n\n #[error(\"downstream response was missing field `{field}`\")]\n MissingField { field: &'static str },\n}\n```\n\n```rust\n// src/client/mod.rs\npub fn fetch_account(&self, id: &str) -> Result<Account, ClientError> {\n let resp = self.http.get(format!(\"{}/accounts/{id}\", self.base)).send()?; // #[from] reqwest::Error\n\n if !resp.status().is_success() {\n return Err(ClientError::Status { status: resp.status().as_u16() });\n }\n\n let bytes = resp.bytes()?;\n let account: Account = serde_json::from_slice(&bytes).map_err(ClientError::Decode)?;\n Ok(account)\n}\n```\n\nKey points, in the order they matter:\n\n**Use `thiserror` for the error enum, one variant per distinct failure a caller might want to branch on.** A library's error type is part of its public API — callers pattern-match on it, so it should name the cases rather than be an opaque string. `#[from]` gives you the `?` conversion for the wrapped source error; `#[source]` links an error you're deliberately not deriving `From` for (avoids an orphan/ambiguous `From` impl when one source type could map to multiple variants).\n\n**Propagate with `?` and never `.unwrap()`/`.expect()` here.** `send()`, `bytes()`, and the decode are all fallible against a network and untrusted bytes — `?` is exactly the tool. `map_err(ClientError::Decode)` is only needed when you can't use `#[from]` (e.g. you want the variant to carry different context, or the `From` would be ambiguous).\n\n**Keep `anyhow` out of the library's public signature.** `anyhow::Result` is for application/binary entry points where you just print the error and exit. Using it in a library strips callers of the ability to match, and it's a semver-breaking change to swap it out later. If a caller wants `anyhow`, they can write `.map_err(anyhow::Error::from)` at *their* boundary — that direction works fine.\n\n**Add `.context(...)` at the boundary where you convert to `anyhow`, not inside the library.** Context strings in the library tend to become either redundant (duplicated in the `#[error]` message) or misleading (loses which call site failed). Let the app layer attach call-site context.\n\n**Decide ownership deliberately.** `&str` for `id` since you only read it; return an owned `Account` since the caller outlives the borrowed `resp` bytes. Don't return a borrow of the response buffer — it won't compile once the buffer is dropped, and `resp.bytes()` consumes the response anyway.\n\nA couple of things to check before you write it:\n\n- Match the crate's declared `edition`/`rust-version` from `Cargo.toml` — don't reach for a newer feature than the MSRV allows.\n- Look at a sibling module: if the crate already has an established error enum or naming convention, extend that rather than introducing a second parallel `ClientError`. One error type per module boundary is the norm; the crate's existing shape wins over my template.\n\nIf this function is `async`, none of the above changes — but make sure the HTTP call is the async client (e.g. `reqwest`'s async `send()` and `.await`), not a blocking one, and don't hold a `std::sync::Mutex` guard across the `.await`.",
1190
+ "outputSha256": "9f5ffdfb82c2e9cb13651ff3b9cb99c2c32d4bb3a3c484c23eced8105dbe3f3f",
1191
+ "promptSha256": "d9370f303af691a8965af367794be498f1f151f3280eb31d994ca08f348dc6b0",
1192
+ "deterministic": [
1193
+ true
1194
+ ],
1195
+ "judge": {
1196
+ "verdict": "pass",
1197
+ "reason": "Answer shows `?` on send()/bytes() and map_err(ClientError::Decode) (criterion 1). Defines thiserror enum with `Request(#[from] reqwest::Error)` and `Decode(#[source] serde_json::Error)` (criterion 2). Explains #[from] gives ? conversion and #[source] links the source error, and that callers pattern-match the concrete type (criterion 3). Warns against .unwrap()/.expect() rather than recommending it."
1198
+ },
1199
+ "passed": true
1200
+ },
1201
+ {
1202
+ "output": "Handle the `Result` by propagating it with `?` at each fallible call, and give the downstream error a home in your crate's own error type rather than leaking the dependency's error type through your public API.\n\n**1. Discover the crate's existing convention first**\n\nBefore writing anything, check:\n- `Cargo.toml` for `edition`/`rust-version` (don't use a feature newer than the MSRV), and whether this is a library or binary target.\n- A neighboring module in the crate: does it already use `thiserror` enums, or `anyhow`? Is there an existing error type the new function should reuse or extend?\n\nFor a **library** crate (which this is), the convention is almost always a `thiserror`-derived error enum — not `anyhow`. `anyhow::Error` is for application/binary entry points where errors are just reported; a library needs a concrete, matchable error type that callers can inspect.\n\n**2. Model the downstream error in your own enum**\n\nWrap the client's error with `#[from]` so `?` converts it automatically, and `#[source]` so the chain is preserved for `Display`/`Error::source`:\n\n```rust\nuse thiserror::Error;\n\n#[derive(Debug, Error)]\npub enum OrderError {\n #[error(\"downstream client call failed\")]\n Client(#[from] downstream::ClientError),\n\n #[error(\"invalid order id: {0}\")]\n InvalidId(#[source] std::num::ParseIntError),\n}\n```\n\nIf the downstream type is internal-only and you don't want it in your public API, use a hand-written `From` impl that maps it into an opaque variant (or a `#[error(\"…\")] Downstream(#[source] Box<dyn std::error::Error + Send + Sync>)`) so the concrete foreign type stays out of your exported surface.\n\n**3. Propagate with `?`, don't unwrap**\n\n```rust\npub fn submit(&self, order_id: &str) -> Result<Receipt, OrderError> {\n let id = order_id.parse::<u64>()?; // InvalidId via From\n let receipt = self.client.submit(id)?; // Client via #[from]\n Ok(receipt)\n}\n```\n\nEvery `?` here is a fallible call on untrusted or otherwise fallible input — never `.unwrap()`/`.expect()` on it. The only place an `expect` is defensible is a genuinely infallible invariant (e.g. a compile-time-known regex or a literal), and even then say why.\n\n**4. Design decisions to make before coding**\n\n- **Ownership of the return type:** return an owned `Result`; don't take a `&mut self` if a `&self` call suffices for the client handle.\n- **`?` conversion cost:** `#[from]` on a wrapping variant gives you automatic conversion; if you need extra context (which order, which retry), don't rely on `#[from]` alone — map explicitly, e.g. `.map_err(|e| OrderError::Client { id, source: e })`.\n- **Async:** if the crate is async, the client call must also be async or routed through `spawn_blocking`; never block the runtime.\n\n**5. Verify**\n\n```bash\ncargo build --all-targets\ncargo clippy --all-targets -- -D warnings\ncargo fmt --check\ncargo test\n```\n\nThen confirm: no `.unwrap()` on the fallible path, the downstream error is represented in your enum (with `#[source]`/`#[from]` as appropriate), and the foreign error type isn't unintentionally in your public API unless you want it there.",
1203
+ "outputSha256": "87cc83904c538bdd08fbbac83afe4550a13665d1ce3f0ffbbf4e115a510b422f",
1204
+ "promptSha256": "d9370f303af691a8965af367794be498f1f151f3280eb31d994ca08f348dc6b0",
1205
+ "deterministic": [
1206
+ true
1207
+ ],
1208
+ "judge": {
1209
+ "verdict": "pass",
1210
+ "reason": "Answer shows `?` propagation in submit(): `self.client.submit(id)?` and `order_id.parse::<u64>()?`, explicitly warns against .unwrap()/.expect(). Names concrete variant `Client(#[from] downstream::ClientError)` and `InvalidId(#[source] std::num::ParseIntError)`. Explains #[source]/#[from] preserve the chain for Display/Error::source so callers can inspect. All criteria met."
1211
+ },
1212
+ "passed": true
1213
+ },
1214
+ {
1215
+ "output": "For a library crate calling a downstream client, the answer depends on whether you're wrapping the client's error or letting it escape, and on your crate's own error-handling style. Work through it in this order.\n\n**1. Match the crate's existing error style first.** Read 1–2 neighboring modules and `Cargo.toml`. If the crate already has an error enum (usually `thiserror`-derived), add a variant to it rather than starting a parallel scheme. If the crate returns `anyhow::Result` everywhere even in library code, that's a signal it isn't meant to be consumed by callers who need to match on error kinds — follow suit. Do not introduce `anyhow` into a library crate that otherwise exposes typed errors: it erases the variants downstream callers need to match on.\n\n**2. Decide whether the client's error is part of your public contract.**\n\n- If your callers may legitimately need to distinguish \"the downstream timed out\" from \"the downstream rejected the request,\" wrap it as a named variant with a source:\n ```rust\n #[derive(Debug, thiserror::Error)]\n pub enum OrderError {\n #[error(\"downstream client rejected the order\")]\n Rejected(#[source] DownstreamError),\n #[error(\"could not reach downstream client\")]\n Unavailable(#[source] DownstreamError),\n }\n ```\n The `#[source]` attribute preserves the error chain so `anyhow`-style reporting at a higher layer still prints the full cause, while your typed callers can still `match` on `OrderError`.\n\n- If the client's error is an implementation detail your callers should never see, use `#[from]` so `?` converts automatically, and keep the variant opaque:\n ```rust\n #[error(\"downstream call failed\")]\n Downstream(#[from] DownstreamError),\n ```\n This lets you write `client.submit(order).await?` with no manual mapping. Prefer `#[from]` when there's exactly one obvious mapping; use `#[source]` with a named variant when you're distinguishing causes.\n\n- If your function's only failure mode *is* the client's error and your crate already treats that error as public API, you can return the client's `Result` type directly instead of wrapping. This is the right call only when you add no error context of your own — the moment you want to distinguish two failure causes from the same call, wrap.\n\n**3. At the call site, propagate with `?`, not `.unwrap()`.** Add context only when your layer knows something the client's error doesn't — e.g. which order was being submitted:\n\n```rust\nlet receipt = client\n .submit(&order)\n .await\n .map_err(|source| OrderError::Rejected(source))?;\n```\n\nReserve `.expect()` for invariants you can prove, and never on input derived from the network or from caller-supplied data. If the response body must be parsed, that parse is fallible too — give it its own variant (`MalformedResponse`) rather than folding it into the transport error.\n\n**4. If the crate is async, check what the client call blocks on.** If the downstream client is itself async, `.await` it directly. If you're calling a blocking client (a synchronous HTTP library, a `std::net` socket), wrap it in `tokio::task::spawn_blocking` rather than calling it inline — a blocking call inside an `async fn` ties up the executor's worker thread for every other task on it. Also make sure no `std::sync::Mutex` guard is held across the `.await`; use `tokio::sync::Mutex` if you need a lock that spans an await point.\n\n**5. Verify before reporting done:**\n- `cargo build --all-targets`, `cargo clippy --all-targets -- -D warnings`, `cargo fmt --check`, `cargo test` all exit 0.\n- The client's error is reachable in the chain (`#[source]`/`#[from]`), not swallowed into a `String` — callers who need it can still find it.\n- No `.unwrap()`/`.expect()` on the client result or on parsing its response.\n- If async: no blocking call inline, no `std::sync` guard held across `.await`.\n\nThe short version: wrap the client error in a variant of your crate's existing error enum, use `#[from]` if you're just passing it through and `#[source]` with a named variant if you're distinguishing causes, propagate with `?`, and don't flatten a typed error into a string unless the crate is uniformly `anyhow`-based.",
1216
+ "outputSha256": "37cf6f8885b57b18b10e9936f625b3b7eb9335201024f90445cfec16d47a7e45",
1217
+ "promptSha256": "d9370f303af691a8965af367794be498f1f151f3280eb31d994ca08f348dc6b0",
1218
+ "deterministic": [
1219
+ true
1220
+ ],
1221
+ "judge": {
1222
+ "verdict": "pass",
1223
+ "reason": "Criterion 1: shows `client.submit(order).await?` and `.map_err(|source| OrderError::Rejected(source))?`, propagating with ?. Criterion 2: names concrete variants `Rejected(#[source] DownstreamError)` and `Downstream(#[from] DownstreamError)`. Criterion 3: explains #[source] preserves the error chain so callers can match/inspect the cause. No fail criterion: .unwrap()/.expect() only warned against."
1224
+ },
1225
+ "passed": true
1226
+ }
1227
+ ]
1228
+ },
1229
+ {
1230
+ "id": "async-blocking-call",
1231
+ "kind": "behavior",
1232
+ "prompt": "I need to write an async fn in a tokio-based Rust service that reads a config file and then sleeps briefly between retries. What should I watch out for?",
1233
+ "strictness": "high",
1234
+ "trials": 10,
1235
+ "passes": 10,
1236
+ "passRate": 1,
1237
+ "passAtK": 1,
1238
+ "grader": "judge",
1239
+ "status": "ran",
1240
+ "trialRecords": [
1241
+ {
1242
+ "output": "The two things to watch out for map directly onto the async-safety rule in this skill:\n\n**1. Use async sleep, not `std::thread::sleep`.**\nInside an `async fn` on a multi-threaded tokio runtime, calling `std::thread::sleep(...)` blocks the worker thread it's running on — every other task scheduled on that thread stalls too, not just yours. Use `tokio::time::sleep(Duration::from_millis(n)).await` instead. Same applies more generally: any blocking call on your path (a synchronous `std::fs` read/socket read, a `std::sync::Mutex` held across an `.await`) can stall the executor. If you genuinely have CPU-bound or unavoidably synchronous work, route it through `tokio::task::spawn_blocking`.\n\n**2. Reading a config file: prefer async I/O, and decide whether an initial read is even fallible-tolerant.**\n- Use `tokio::fs::read_to_string(path).await` (or the async equivalent appropriate to the format) rather than `std::fs::read_to_string`, so the read doesn't block the runtime. A config file read is I/O-bound, so it belongs on the async path.\n- Propagate the open/read error with `?` — don't `.unwrap()` it. A missing or malformed config file is exactly the \"fallible input\" case; surface it as an error (with `.context(...)` if this is a binary/app entry point using `anyhow`, or a `thiserror` variant carrying `#[source]`/`#[from]` if this is a library module).\n\n**A few smaller things worth deciding up front:**\n\n- **Retry loop shape.** Put the `tokio::time::sleep(...).await` between attempts, and make sure the retry bound is explicit (attempt count or deadline) rather than an unbounded loop. If retries can run long, consider whether the config load itself belongs on a timeout.\n- **Parse-then-validate.** If the config is deserialized (serde), parse errors and validation errors are distinct failure modes; give each a sane error variant rather than collapsing them into a string.\n- **Formatting/borrowing.** Take `&Path`/`&str` for the path argument rather than owning a `String` if the caller can lend it; that's the default per the coding-style rule, not a reflexive `.clone()`.\n- **Edition/MSRV.** Check `Cargo.toml` for the declared `edition` and `rust-version` before reaching for newer language features.\n\n**Verify before calling it done:** `cargo build --all-targets`, `cargo clippy --all-targets -- -D warnings`, `cargo fmt --check`, and `cargo test` all exit 0; no `.unwrap()`/`.expect()` on the config read; and no blocking call introduced on the async path without `spawn_blocking` or an async equivalent.",
1243
+ "outputSha256": "22eb0c3526b7fe0675483e51c5bc6ab45b49f03d309b60260085915f668434db",
1244
+ "promptSha256": "c7f9f4faf4b75e041af6e5ed8faebd7745c4877596ed9eeffb26025abee596cb",
1245
+ "deterministic": [],
1246
+ "judge": {
1247
+ "verdict": "pass",
1248
+ "reason": "Criterion 1: names tokio::fs::read_to_string(path).await instead of std::fs::read_to_string. Criterion 2: names tokio::time::sleep(Duration::from_millis(n)).await instead of std::thread::sleep. Criterion 3: explains std::thread::sleep blocks the worker thread so 'every other task scheduled on that thread stalls too, not just yours.' Also routes blocking work through tokio::task::spawn_blocking. No fail criteria triggered."
1249
+ },
1250
+ "passed": true
1251
+ },
1252
+ {
1253
+ "output": "The main thing to watch out for: **none of the obvious ways to read a file or sleep are safe to call directly inside `async fn` on a multi-threaded tokio runtime.**\n\n## Sleeping between retries\n\n`std::thread::sleep` blocks the OS thread the task is running on. On a multi-threaded runtime that is one of the runtime's worker threads, so every other task scheduled on that thread stalls for the duration — not just yours. Use `tokio::time::sleep(Duration::from_millis(n)).await` instead. If you want jittered backoff between retries rather than a flat sleep, `tokio::time::sleep` composes with any jitter you compute yourself; there is no built-in jitter helper.\n\n## Reading the config file\n\nThis is the part people get wrong more often than the sleep, because a plain synchronous read *looks* harmless:\n\n- `std::fs::read_to_string(path)?` inside `async fn` is a blocking syscall. For a small file it usually finishes fast enough that nobody notices, until it lands on a network-mounted path, a cold page cache, or a slow disk — then it stalls a worker thread.\n- Use `tokio::fs::read_to_string(path).await?`. Under the hood tokio routes file I/O through a blocking thread pool, so it does not stall the executor, and it yields cleanly at the `.await`.\n- If you're not using `tokio::fs`, the alternative is `tokio::task::spawn_blocking(move || std::fs::read_to_string(path))` and `.await`ing the join handle. That keeps the blocking call off the executor thread.\n- If the config is parsed with a synchronous parser (serde, etc.) after the read, that's fine — parsing an in-memory string is CPU-bound and non-blocking, so it does not need `spawn_blocking` unless the file is large enough to be a real CPU cost.\n\n## The retry loop itself\n\nShape it so the blocking concern is handled at each step:\n\n```rust\nuse std::time::Duration;\n\nasync fn load_config(path: &std::path::Path, retries: u32) -> Result<Config, ConfigError> {\n let mut attempt = 0;\n loop {\n match tokio::fs::read_to_string(path).await {\n Ok(raw) => return parse_config(&raw).map_err(ConfigError::from),\n Err(e) if attempt < retries => {\n attempt += 1;\n tokio::time::sleep(backoff(attempt)).await;\n }\n Err(e) => return Err(ConfigError::from(e)),\n }\n }\n}\n```\n\nA few things this shape gets right that are easy to miss:\n\n- **Every `.await` point is a cancellation point.** If the caller wraps this in `tokio::time::timeout` or the task is aborted, cancellation can land between the read and the sleep. That's usually fine, but don't hold a lock or a partially-mutated buffer across an `.await` if the function may be cancelled — the state won't be unwound the way you might assume.\n- **Don't hold a `std::sync::Mutex` guard across the `.await`.** A `std::sync::MutexGuard` is not `Send`, so the future won't compile if it needs to be `Send` — or worse, if you work around it, you can block the executor waiting on a lock another task holds across its own await. Use `tokio::sync::Mutex` if the lock genuinely must span an await, or restructure to drop the guard before awaiting.\n- **Backoff should actually grow.** A flat retry loop with a constant sleep is a hammer; exponential backoff with jitter is the usual default for retrying file or network reads.\n- **Retrying a missing file may be pointless.** If the error is `ErrorKind::NotFound` and nothing is going to create the file, retrying just delays the inevitable — consider matching on the error kind and only retrying the transient ones (busy, interrupted, network-backed I/O errors).\n\n## MSRV and edition\n\nCheck `Cargo.toml` for the pinned `rust-version` and `edition` before reaching for anything newer than what's declared. `tokio::task::spawn_blocking` and `tokio::fs` have been stable for a long time, so this is rarely the constraint, but confirm rather than assume — especially if the crate is pinned to an older edition.\n\n## Bottom line\n\nTwo rules cover the whole question: never call `std::thread::sleep` or a synchronous `std::fs` operation inside `async fn` on a multi-threaded runtime. Use `tokio::time::sleep` and either `tokio::fs` or `spawn_blocking`. Everything else — the retry shape, cancellation, lock handling — follows from those two.\n\nBefore you call it done: `cargo clippy --all-targets -- -D warnings` will not flag a `std::thread::sleep` or blocked `std::fs` call on its own, so verify by reading the function for synchronous I/O and sleeps rather than trusting the linter to catch it.",
1254
+ "outputSha256": "8c8ec5efa2337e1e35e89a8a2f9aa3a9c62671de7c75cd8a816a4f7f6473b3ec",
1255
+ "promptSha256": "c7f9f4faf4b75e041af6e5ed8faebd7745c4877596ed9eeffb26025abee596cb",
1256
+ "deterministic": [],
1257
+ "judge": {
1258
+ "verdict": "pass",
1259
+ "reason": "Criterion 1: names tokio::fs::read_to_string(path).await? and warns std::fs::read_to_string blocks. Criterion 2: names tokio::time::sleep(Duration::from_millis(n)).await instead of std::thread::sleep. Criterion 3: explains std::thread::sleep stalls a worker thread so 'every other task scheduled on that thread stalls for the duration — not just yours.' Also offers spawn_blocking. No fail criteria met."
1260
+ },
1261
+ "passed": true
1262
+ },
1263
+ {
1264
+ "output": "The two classic traps here are both in your sentence: file reads and sleeps inside an `async fn` on a multi-threaded Tokio runtime.\n\n**1. The config file read is a blocking call — don't call it inline.**\n\n`std::fs::read_to_string` (and `std::fs::metadata`, directory walks, etc.) block the calling OS thread. On a multi-threaded Tokio runtime that thread is a worker owned by the executor, so a slow or cold-cache read stalls every other task scheduled on that worker, not just yours. Use the async equivalent:\n\n```rust\nlet contents = tokio::fs::read_to_string(path).await?;\n```\n\nIf the path only ever resolves to a tiny file you read once at startup, a blocking read before the runtime is fully in service can be defensible — but inside a running async service, use `tokio::fs` or route it through `tokio::task::spawn_blocking`. Note `tokio::fs` is itself backed by `spawn_blocking`, so it isn't free; if you're reading the same config repeatedly, cache it rather than re-reading per retry.\n\n**2. The \"brief sleep\" must be `tokio::time::sleep`, not `std::thread::sleep`.**\n\n`std::thread::sleep` parks the worker thread for the whole duration — same executor-stall problem, and it's especially insidious here because a retry backoff sleep is exactly where it hurts most (you're already in a degraded path). Use:\n\n```rust\ntokio::time::sleep(Duration::from_millis(backoff)).await;\n```\n\nRequires the `time` feature on your `tokio` dependency. Consider `tokio::time::timeout` around the whole retry loop so a hung attempt can't run forever.\n\n**Related pitfalls on this same path:**\n\n- **Don't hold a `std::sync::Mutex`/`RwLock` guard across the `.await`.** If the config is behind a `std::sync::Mutex` and you lock, then `.await` on the file read or the sleep, you can deadlock or block a worker. Either scope the lock so it's dropped before any `.await`, or use `tokio::sync::Mutex` if it genuinely must be held across an await. Guarding a plain config value that you clone out is usually better still.\n- **Error handling on retries.** Propagate with `?` / `anyhow::Result` with `.context(...)`, not `.unwrap()`. A config file that's missing or malformed is fallible input — a panic here takes down the whole service. Distinguish retryable errors (transient I/O) from fatal ones (the file doesn't exist / fails to parse) so you don't burn the retry budget on a hopeless case.\n- **Backoff growth.** A fixed \"brief sleep\" retried in a tight loop can turn into a hot loop or a thundering herd. Prefer exponential backoff with jitter, and cap the total attempts.\n- **Cancellation.** Any `.await` point is a cancellation point. If this runs inside a `select!` or a task that can be dropped, make sure dropping mid-retry leaves no half-initialized state — the config value should only be \"committed\" once fully read and parsed.\n- **Edition / MSRV.** Match whatever `edition` and `rust-version` your `Cargo.toml` declares before reaching for newer syntax.\n\nSketch of the safe shape:\n\n```rust\nasync fn load_with_retry(path: &Path, attempts: u32) -> anyhow::Result<Config> {\n let mut delay = Duration::from_millis(100);\n for attempt in 0..attempts {\n match tokio::fs::read_to_string(path).await {\n Ok(s) => return Ok(parse(&s)?), // fatal on parse error -> propagate\n Err(e) if attempt + 1 == attempts => return Err(e).context(\"reading config\"),\n Err(_) => {\n tokio::time::sleep(delay).await; // async sleep, not thread::sleep\n delay = (delay * 2).min(Duration::from_secs(5));\n }\n }\n }\n unreachable!()\n}\n```\n\nThen verify with `cargo clippy --all-targets -- -D warnings` — though note clippy won't catch a stray `std::thread::sleep` for you, so the blocking-call discipline is on you (or on a review) rather than the toolchain.",
1265
+ "outputSha256": "4012637c3534a0ebf3a4723edb48abc9fa1360b27d0b49df605e0ba957358c47",
1266
+ "promptSha256": "c7f9f4faf4b75e041af6e5ed8faebd7745c4877596ed9eeffb26025abee596cb",
1267
+ "deterministic": [],
1268
+ "judge": {
1269
+ "verdict": "pass",
1270
+ "reason": "Names tokio::fs::read_to_string (code block and sketch) instead of std::fs; names tokio::time::sleep instead of std::thread::sleep (code and sketch). Explains concrete consequence: blocking read 'stalls every other task scheduled on that worker, not just yours'; sleep 'parks the worker thread... same executor-stall problem'. Also routes via spawn_blocking. No fail criteria triggered."
1271
+ },
1272
+ "passed": true
1273
+ },
1274
+ {
1275
+ "output": "The core hazard: **`std::fs::read` and `std::thread::sleep` are both blocking**, and calling either directly inside an `async fn` on tokio's multi-threaded runtime stalls a worker thread — every other task scheduled on that thread waits too, not just this one.\n\n**The config read** — use `tokio::fs::read_to_string` instead of `std::fs::read_to_string`. The async equivalent is a drop-in for this use case. Note it's not a free lunch internally (tokio backs it with `spawn_blocking` under the hood), but it yields the worker thread back to the executor while the read is in flight, which is the point. If the file is small and read once at startup, you can also just do it before the async runtime is doing any concurrent work, but the async version is still the right default.\n\n**The sleep between retries** — use `tokio::time::sleep`, not `std::thread::sleep`. Same reasoning. If you want backoff, `tokio::time::sleep` composes with a checked-in `Duration` argument. Don't hand-roll with a busy loop.\n\n**Three more that bite specifically in retry code:**\n\n1. **Don't hold a `std::sync::Mutex` (or an `RwLock`) guard across an `.await`.** The guard isn't `Send`-safe across the yield point; the compiler normally catches this with a `future is not Send` error, but if you're in a `!Send` context or it slips through, you get a deadlock or a data-race-shaped bug. Use `tokio::sync::Mutex` if the lock genuinely must be held across await points, and prefer not to design that way if you can avoid it.\n\n2. **Propagate the file error with `?`, don't `.unwrap()` it.** A missing or permission-denied config file is exactly the kind of fallible, environment-dependent input the unwrap rule targets. Give the `io::Error` context and propagate.\n\n3. **Bound the retry loop.** An unbounded `loop { ... sleep ... }` around a read that keeps failing is a silent hang. Cap attempts or total elapsed time and return a real error when exhausted.\n\nSketch of the shape:\n\n```rust\nuse tokio::time::{sleep, Duration};\n\nasync fn load_config_with_retry(path: &str, attempts: u32) -> Result<String, ConfigError> {\n let mut last_err = None;\n for attempt in 0..attempts {\n match tokio::fs::read_to_string(path).await {\n Ok(contents) => return Ok(contents),\n Err(e) => {\n last_err = Some(e);\n if attempt + 1 < attempts {\n sleep(Duration::from_millis(100 << attempt)).await; // exponential-ish backoff\n }\n }\n }\n }\n Err(ConfigError::Exhausted {\n attempts,\n source: last_err.expect(\"attempts >= 1 guarantees at least one error\"),\n })\n}\n```\n\n(`attempts` must be `>= 1` for that `expect` to hold — either assert it at the call boundary or restructure to avoid the `expect` entirely; the rule against `.unwrap()`/`.expect()` still applies here, this is only defensible because the invariant is stated and checked at the boundary.)\n\nFinally, match the crate's declared `edition`/`rust-version` in `Cargo.toml` — don't reach for a newer async language feature than what's pinned. And run `cargo clippy --all-targets -- -D warnings` before calling it done; clippy's `await_holding_lock` and related lints catch several of the above mechanically.",
1276
+ "outputSha256": "84359a9d458b63b774bec83495a9586d9a7f8ce6bcd4df828a2f72229239ab2d",
1277
+ "promptSha256": "c7f9f4faf4b75e041af6e5ed8faebd7745c4877596ed9eeffb26025abee596cb",
1278
+ "deterministic": [],
1279
+ "judge": {
1280
+ "verdict": "pass",
1281
+ "reason": "Names tokio::fs::read_to_string instead of std::fs (config read) and tokio::time::sleep instead of std::thread::sleep (retry delay). Explains the concrete consequence: blocking stalls a worker thread so 'every other task scheduled on that thread waits too, not just this one.' No fail criteria triggered."
1282
+ },
1283
+ "passed": true
1284
+ },
1285
+ {
1286
+ "output": "Two things in that plan are blocking calls that do not belong in an `async fn` on a multi-threaded tokio runtime, and they are the ones to watch for.\n\n**1. Reading the config file**\n\n`std::fs::read` / `std::fs::read_to_string` (and `File::open` + `read`) are synchronous, blocking syscalls. Inside an `async fn` on a multi-thread runtime, a blocking call parks the executor worker thread it is running on for the duration of the syscall — every other task scheduled on that worker is stalled too, not just yours. A config file is usually small so the stall is usually short, but \"usually short\" is not a guarantee (network-mounted filesystem, cold cache, slow disk).\n\nOptions, best first for a config file:\n\n- Use the async equivalent: `tokio::fs::read_to_string(path).await`. This is the direct, idiomatic answer.\n- If the read genuinely must be synchronous (e.g. a library API that only exposes a blocking read), route it through `tokio::task::spawn_blocking(move || ...).await` so it runs on the dedicated blocking pool rather than a worker thread.\n\nDo not hold a `std::sync::Mutex` (or any `!Send`/blocking guard) across the `.await` for the read either — that is the other classic way to stall or fail to compile on a multi-thread runtime. If the config needs to be shared, prefer `tokio::sync::RwLock`/`Mutex`, or read it once into an owned `Arc<T>` and share that.\n\n**2. The sleep between retries**\n\n`std::thread::sleep(Duration)` inside an `async fn` blocks the worker thread for the entire retry delay — the worst version of the same problem, because retry backoff can be seconds. Use:\n\n```rust\ntokio::time::sleep(Duration::from_millis(backoff)).await;\n```\n\nAlso check `Cargo.toml` for the `tokio` feature set — `tokio::fs` and `tokio::time` require the `fs` and `time` features respectively; if they are not enabled, enable them rather than falling back to the `std` versions.\n\n**A few adjacent things worth deciding before you write it:**\n\n- **Backoff shape.** For \"sleeps briefly between retries,\" consider exponential backoff with a cap (`tokio::time::sleep(base * 2u32.pow(attempt))`) and, for anything talking to a real service, jitter so retries from many callers do not synchronize.\n- **Error propagation.** The read and the attempt you are retrying are both fallible. Propagate with `?` where the failure should end the loop, and only continue the retry loop on the specific error variants that are retryable. Do not `.unwrap()` the config read — a missing or malformed config file is exactly the \"should always be valid\" case that panics the process; return a `Result` (a `thiserror` enum for a library module, `anyhow::Result` with `.context(...)` at a binary boundary) instead.\n- **Read the config once, not per retry.** If the config is static, read it before the retry loop. Re-reading a file on every attempt inside the loop triples the number of blocking-risk points for no benefit, and the file could change mid-loop.\n- **Retry bound.** An unbounded retry loop that sleeps is a hang risk; cap the attempts (or apply an overall deadline via `tokio::time::timeout`) and surface exhaustion as an error variant.\n\nSo the shape to aim for is roughly: `tokio::fs::read_to_string(...).await?` once, then a bounded loop whose body `.await`s the real attempt and whose backoff is `tokio::time::sleep(...).await` — with no `std::fs`, `std::thread::sleep`, or `std::sync` guard held across an `.await` anywhere on the path.",
1287
+ "outputSha256": "727b7684ea985b10fe5a39e17d58e175b9b1662283d99f66bd0c9775cb621e7d",
1288
+ "promptSha256": "c7f9f4faf4b75e041af6e5ed8faebd7745c4877596ed9eeffb26025abee596cb",
1289
+ "deterministic": [],
1290
+ "judge": {
1291
+ "verdict": "pass",
1292
+ "reason": "Criterion 1: names tokio::fs::read_to_string(path).await as the direct idiomatic answer. Criterion 2: names tokio::time::sleep(Duration::from_millis(backoff)).await. Criterion 3: explains blocking parks the executor worker thread, stalling every other task scheduled on that worker, not just yours. Also routes unavoidable blocking through spawn_blocking. No fail criteria met."
1293
+ },
1294
+ "passed": true
1295
+ },
1296
+ {
1297
+ "output": "Watch out for the two blocking traps — this pattern hits both.\n\n## 1. Don't read the config file with `std::fs`\n\n`std::fs::read_to_string` / `File::open` are **synchronous** blocking calls. Inside an `async fn` on a multi-threaded tokio runtime, a blocking call stalls the *worker thread*, not just your task — every other task scheduled on that thread waits behind it. Use the async equivalent:\n\n```rust\n// bad inside async fn on a tokio multi-thread runtime\nlet s = std::fs::read_to_string(path)?;\n\n// good\nlet s = tokio::fs::read_to_string(path).await?;\n```\n\nIf you have to call a sync API you can't replace (e.g. a blocking parser, a sync third-party client), route it through the blocking pool:\n\n```rust\nlet s = tokio::task::spawn_blocking(move || std::fs::read_to_string(path))\n .await\n .context(\"config read task panicked\")??;\n```\n\nNote the double `?`: the outer `JoinError` (the task panicked/was cancelled) and the inner `io::Result`. Don't collapse them into one `.unwrap()`.\n\n## 2. Don't `std::thread::sleep` between retries\n\n`std::thread::sleep` blocks the worker thread for the whole duration. Use the async sleep:\n\n```rust\ntokio::time::sleep(Duration::from_millis(backoff)).await;\n```\n\n## 3. Everything else in that async fn\n\n- **Anything sync in the retry loop** — a sync HTTP client, a sync DB driver, `std::sync::Mutex` held across an `.await`. That last one is subtle: if you `.lock()` a `std::sync::Mutex`, then `.await` while still holding the guard, another task on the same thread can deadlock or the future won't be `Send`. Hold the guard in a tight scope, or use `tokio::sync::Mutex` if you genuinely must hold across `.await`.\n- **Error propagation** — this is fallible I/O and, if the config is user-supplied, untrusted input. Propagate with `?`, don't `.unwrap()`. If the config read is *supposed* to be optional, that's an explicit `Option`/`match`, not an `.expect()` on the happy path.\n- **Retry policy** — cap attempts, add jitter/backoff, and make sure the config file is re-read (or the old value dropped) per attempt if that's the intent; otherwise you're retrying a stale value.\n\n## Sketch\n\n```rust\nasync fn load_config_with_retries(path: &Path) -> Result<Config, ConfigError> {\n let mut attempt = 0;\n loop {\n match tokio::fs::read_to_string(path).await {\n Ok(s) => return Ok(toml::from_str(&s)?), // parse\n Err(e) if attempt < MAX_RETRIES => {\n attempt += 1;\n let backoff = backoff_with_jitter(attempt); // cap + jitter\n tokio::time::sleep(backoff).await; // async sleep, not thread::sleep\n }\n Err(e) => return Err(ConfigError::Read { source: e, attempts: attempt }),\n }\n }\n}\n```\n\n## Before you call it done\n\n`cargo clippy --all-targets -- -D warnings` catches some of this (e.g. blocking-in-async annotations), but not all of it — the `std::sync::Mutex`-across-`.await` case in particular needs a read-through by eye. Grep the new path for `std::fs`, `std::thread::sleep`, `std::net`, `reqwest::blocking`, and any `.lock()` on a `std::sync` primitive, and confirm each one is either gone or wrapped in `spawn_blocking`/scoped away from the `.await`.",
1298
+ "outputSha256": "aea0c7b0917d2bafc19013d825eb402018cbd1b7cc2ddc8830486c0e7270760f",
1299
+ "promptSha256": "c7f9f4faf4b75e041af6e5ed8faebd7745c4877596ed9eeffb26025abee596cb",
1300
+ "deterministic": [],
1301
+ "judge": {
1302
+ "verdict": "pass",
1303
+ "reason": "Criterion 1: names tokio::fs::read_to_string as the good replacement for std::fs::read_to_string. Criterion 2: names tokio::time::sleep(Duration::from_millis(backoff)).await instead of std::thread::sleep. Criterion 3: explains a blocking call 'stalls the worker thread, not just your task — every other task scheduled on that thread waits behind it.' Also routes sync APIs through tokio::task::spawn_blocking. No fail criteria hold."
1304
+ },
1305
+ "passed": true
1306
+ },
1307
+ {
1308
+ "output": "The two traps in exactly what you described:\n\n**1. Blocking file read inside an `async fn`.** `std::fs::read_to_string` (and all of `std::fs`) is a synchronous syscall. Called directly inside an `async fn` on a multi-threaded Tokio runtime, it stalls the executor's worker thread that happens to be running your future — which is also running every other task scheduled on that thread. A slow or contended read doesn't just slow down your retry loop; it freezes unrelated requests. Use the async equivalent — `tokio::fs::read_to_string`, with `tokio`'s `fs` feature enabled in `Cargo.toml` — or, if you must call synchronous code, wrap it in `tokio::task::spawn_blocking` so it runs on the blocking pool instead of a worker thread.\n\n**2. `std::thread::sleep` between retries.** Same root cause, more blatant: `std::thread::sleep` parks the worker thread for the full duration, holding up every other task on it. Use `tokio::time::sleep(Duration::from_millis(...)).await`. This is the most common version of the bug because `thread::sleep` looks correct in isolation and compiles fine inside an `async fn`.\n\n**Also worth checking while you're there:**\n\n- A lock held across an `.await`. If the retry loop touches shared state under a `std::sync::Mutex` or `RwLock` and you `.await` (the sleep, the file read) while holding the guard, you block the same worker thread and can deadlock. Use `tokio::sync::Mutex`/`RwLock`, or restructure so the guard is dropped before the await point.\n- The retry loop itself should usually be bounded (a max attempt count) with backoff (exponential and/or jittered) rather than a fixed sleep, and should give up with a propagated error via `?` or `.context(...)` — not loop forever on a file that will never appear.\n- Don't `.unwrap()` the file read \"since the config should always be there.\" A missing or malformed config file is exactly the fallible input this rule is about; return a `Result` and let the caller decide whether a missing config is fatal or means \"use defaults.\" If the config is read once at startup, note that startup is often the one place a blocking read is acceptable — but only if you're genuinely not yet serving tasks.\n- If the caller retries the whole function rather than an inner loop, the same two rules apply at that call site; a blocking sleep doesn't stop being blocking because it's one frame up.\n\nThe short version: inside an `async fn` on a multi-threaded runtime, *anything* that can park the thread — sync I/O, `thread::sleep`, a std lock held across an await — is the thing to hunt for, and `spawn_blocking` or the `tokio::` equivalent is the fix.",
1309
+ "outputSha256": "760bcae97b2e5b854f83307d308c9514aeaeac13ec76ba9bede7b50daeed82c7",
1310
+ "promptSha256": "c7f9f4faf4b75e041af6e5ed8faebd7745c4877596ed9eeffb26025abee596cb",
1311
+ "deterministic": [],
1312
+ "judge": {
1313
+ "verdict": "pass",
1314
+ "reason": "Criterion 1: names tokio::fs::read_to_string instead of std::fs. Criterion 2: names tokio::time::sleep(Duration::from_millis(...)).await instead of std::thread::sleep. Criterion 3: explains it stalls the executor's worker thread running every other task on that thread, freezing unrelated requests. Also routes sync code through tokio::task::spawn_blocking. No fail criteria hold."
1315
+ },
1316
+ "passed": true
1317
+ },
1318
+ {
1319
+ "output": "`async fn` on a tokio service, reading a config file, then sleeping between retries — a few things will bite you if you are not deliberate about them. This is scope for `rust-implementation`, drawing on `rules/patterns.mdc`, `rules/coding-style.mdc`, and `rules/security.mdc`, so read those before you write the code.\n\n**The blocking-call trap (the big one).** `std::fs::read_to_string` is a synchronous, blocking call. Dropped inside an `async fn` running on tokio's multi-threaded runtime, it blocks the executor worker thread that is holding the task — every other task scheduled on that same worker stalls until the read returns. A config read on a cold page cache is fast, which is exactly why this bug hides: it only hurts under load or when the file lives on slow/network storage. Either use the async equivalent (`tokio::fs::read_to_string`) or, if you must stay on `std::fs`, route it through `tokio::task::spawn_blocking`. `tokio::fs` is itself a thin `spawn_blocking` wrapper, so either path is correct; a bare `std::fs::read_to_string(...)` mid-`async fn` is not.\n\n**The sleep.** Use `tokio::time::sleep`, not `std::thread::sleep`. `std::thread::sleep` in an `async fn` is the same class of mistake: it parks the whole worker thread instead of yielding the task, and it is invisible to the runtime's timer. If you pull in `tokio`'s time feature and the paused-clock test utilities, note that `tokio::time::sleep` only advances under a `tokio::time::pause()`/`advance()` controlled clock — a bare `#[test]` will not drive it.\n\n**Anything you `.await` across a lock.** If the config is fenced by a `Mutex`, `std::sync::Mutex` guards are not held across `.await`: the guard is not `Send` in a way that survives suspension, and holding it across a yield point deadlocks the executor rather than the task. Hold the guard only long enough to extract the value, or use `tokio::sync::Mutex` if the critical section genuinely spans an await.\n\n**Ownership and the config value.** If the read produces a `String`, decide early whether the caller needs to own it or can borrow. `tokio::fs::read_to_string` gives you an owned `String`; if only part of the parsed config is used downstream, borrow `&str` from it rather than cloning per retry. Before reaching for `.clone()` to end an ownership fight, look at whether the value can be borrowed or moved.\n\n**Error handling on the read.** `fs::read_to_string` returns `Result<_, io::Error>`. Propagate it with `?` — do not `.unwrap()` or `.expect()` it. A config file is untrusted input: it may be missing, unreadable, or racing a deploy that replaces it mid-read. If this is a library module, wrap the `io::Error` in a `thiserror` enum with a `#[from]` variant; if it is the binary entry point, use `anyhow::Result` and add `.context(\"reading config from ...\")` so a failure names the path. Never panic the runtime because a config file was not there.\n\n**Retry shape itself.** Bound the retries. An unbounded `loop { ... tokio::time::sleep(...).await }` with no cap turns a transient failure into a hang. Use a fixed attempt count or an exponential backoff with a ceiling, and decide whether the read failure is retryable (transient I/O) versus terminal (bad parse, missing required field) — retrying a malformed config just delays the inevitable error. Do not read the file once outside the loop and retry the sleep: if the point of retrying is to pick up a config that a deploy is still writing, the read belongs inside the loop.\n\n**Pure-parsing vs I/O.** Parsing the config text (serde, etc.) is CPU-bound and synchronous. That is fine to do inline — it does not block, because it does not await — but if the parse is expensive, get the bytes on the async path and parse off the `?` on owned data. Do not mix the two concerns in a way that hides which part is fallible.\n\nA sketch that respects all of this:\n\n```rust\nuse std::time::Duration;\nuse anyhow::{Context, Result};\n\nasync fn load_config_with_retry(path: &str, attempts: u32) -> Result<Config> {\n let mut last_err = None;\n for attempt in 0..attempts {\n match tokio::fs::read_to_string(path).await {\n Ok(text) => return parse_config(&text).context(\"parsing config\"),\n Err(e) => {\n last_err = Some(anyhow::Error::from(e).context(\"reading config file\"));\n if attempt + 1 < attempts {\n tokio::time::sleep(Duration::from_millis(100 * (attempt + 1) as u64)).await;\n }\n }\n }\n }\n Err(last_err.unwrap_or_else(|| anyhow::anyhow!(\"no attempts made\")))\n}\n```\n\nNote the `?` propagation on the parse, the `tokio::fs` read, `tokio::time::sleep`, the bounded loop, and the read inside the loop.\n\n**Before you call it done:** `cargo build --all-targets`, `cargo clippy --all-targets -- -D warnings`, `cargo fmt --check`, and `cargo test` all clean; no `.unwrap()`/`.expect()` on the fallible read; no blocking call inside the `async fn` without `spawn_blocking`; and only `tokio::time::sleep` between retries.",
1320
+ "outputSha256": "d65da9c5d7892373677dc16f18dee57a0a0931c6d528a861c5d2538a82ec8bf7",
1321
+ "promptSha256": "c7f9f4faf4b75e041af6e5ed8faebd7745c4877596ed9eeffb26025abee596cb",
1322
+ "deterministic": [],
1323
+ "judge": {
1324
+ "verdict": "pass",
1325
+ "reason": "Names tokio::fs::read_to_string (\"Either use the async equivalent (tokio::fs::read_to_string)\" and in the sketch) and tokio::time::sleep (\"Use tokio::time::sleep, not std::thread::sleep\"). Explains the concrete multi-threaded consequence: std::fs \"blocks the executor worker thread... every other task scheduled on that same worker stalls\"; sleep \"parks the whole worker thread\". Also routes blocking work via tokio::task::spawn_blocking. No fail criteria: std::thread::sleep only warned against; no small-file dismissal."
1326
+ },
1327
+ "passed": true
1328
+ },
1329
+ {
1330
+ "output": "In a tokio-based service, the two things you named are exactly the two classic traps — one blocking I/O call, one blocking sleep. Here's what to watch for:\n\n## 1. Don't use blocking file I/O inside the async fn\n\n`std::fs::read_to_string` (or any `std::fs::` call) is synchronous and blocks the calling OS thread for the whole duration of the read. In an `async fn` on tokio's multi-threaded runtime, that stalls the worker thread the task was scheduled on — and every *other* task queued to that same worker, not just yours. On a busy service one slow config read can add latency to unrelated requests.\n\nUse the async equivalent:\n\n```rust\n// async, non-blocking\nlet contents = tokio::fs::read_to_string(&path).await?;\n```\n\nIf for some reason you must use a blocking API (e.g. a library with no async variant), wrap it explicitly so tokio moves it off the worker threads:\n\n```rust\nlet contents = tokio::task::spawn_blocking(move || {\n std::fs::read_to_string(&path)\n})\n.await??; // outer ? for the JoinError, inner ? for the io::Result\n```\n\nNote the double `?` — `spawn_blocking` returns `Result<T, JoinError>`, and your closure returns its own `Result`, so you unwrap both layers. Also note `path` must be moved (or borrowed with a `'static`-friendly lifetime) into the closure.\n\n## 2. Never use `std::thread::sleep` between retries\n\n`std::thread::sleep(Duration::from_millis(200))` blocks the executor thread for 200ms — the same problem as above, and it's very easy to reach for by habit. Use:\n\n```rust\ntokio::time::sleep(Duration::from_millis(200)).await;\n```\n\nThis yields the task back to the executor, so other tasks make progress during the wait. This is the single most common bug in retry loops.\n\n## 3. Don't hold a `std::sync::Mutex` guard across `.await`\n\nIf the retry loop touches shared state, be careful which mutex you use. Holding a `std::sync::MutexGuard` across an `.await` point can deadlock the runtime (the guard isn't `Send`-friendly across suspension, and another task on the same thread may need that lock). Either:\n\n- scope the lock so it's dropped *before* the `.await`:\n ```rust\n {\n let mut guard = state.lock().unwrap();\n guard.attempts += 1;\n } // guard dropped here, before the sleep\n tokio::time::sleep(...).await;\n ```\n- or use `tokio::sync::Mutex` if you genuinely need to hold the lock across an await.\n\n## 4. Shape the error path with `?`, not `.unwrap()`\n\nA config file read is fallible and the path may come from outside — `.unwrap()` panics the task (and, depending on your setup, possibly the whole service). Propagate:\n\n```rust\nasync fn load_config(path: &Path) -> Result<Config, ConfigError> {\n let raw = tokio::fs::read_to_string(path)\n .await\n .map_err(ConfigError::Read)?; // io::Error -> your error\n let parsed = parse(&raw)?; // parse error propagates\n Ok(parsed)\n}\n```\n\nFor a library-level module, a `thiserror` enum with a `#[from]`/`#[source]` on the read variant is the tidy choice; at a binary entry point, `anyhow::Result` with `.context(\"loading config\")` on the `?` is fine.\n\n## 5. Consider whether a retry loop is even the right shape\n\nIf you're retrying the config load because it might fail transiently, that's usually a *startup* concern — typically you load config once before serving traffic rather than retrying inside a hot path. If it really is a runtime reload-with-retry, make the retry count and delay bounded and explicit:\n\n```rust\nasync fn load_with_retry(path: &Path, attempts: u32) -> Result<Config, ConfigError> {\n let mut last_err = None;\n for i in 0..attempts {\n match load_config(path).await {\n Ok(cfg) => return Ok(cfg),\n Err(e) => {\n last_err = Some(e);\n if i + 1 < attempts {\n tokio::time::sleep(Duration::from_millis(200)).await;\n }\n }\n }\n }\n Err(last_err.expect(\"attempts > 0\"))\n}\n```\n\nBeware the `.expect` there — it's only sound because the loop body guarantees `last_err` is `Some` when `attempts >= 1`. If `attempts` can be `0`, that's a panic on an untrusted/edge input; guard it (return an error for `attempts == 0` up front) rather than relying on the caller.\n\n## 6. Respect the crate's declared edition/MSRV\n\nRead `Cargo.toml` before writing — if `tokio` isn't already a dependency, confirm the runtime flavor (multi-thread vs current-thread) and whether `tokio::fs` is even enabled (the `fs` feature may need adding). Don't reach for a language feature newer than the pinned `rust-version`.\n\n---\n\n**The three-line summary:** use `tokio::fs` (or `spawn_blocking`) instead of `std::fs`, use `tokio::time::sleep` instead of `std::thread::sleep`, and don't hold a `std::sync::Mutex` guard across `.await`. Then verify with `cargo clippy --all-targets -- -D warnings` — clippy's `await_holding_lock` lint will catch trap #3 for you.",
1331
+ "outputSha256": "4f30767fd48c14a073d5a68b9622a3f49a9bfd28cd3c176227c05f283ff8f49c",
1332
+ "promptSha256": "c7f9f4faf4b75e041af6e5ed8faebd7745c4877596ed9eeffb26025abee596cb",
1333
+ "deterministic": [],
1334
+ "judge": {
1335
+ "verdict": "pass",
1336
+ "reason": "Answer names tokio::fs::read_to_string (section 1 code) instead of std::fs, names tokio::time::sleep (section 2 code) instead of std::thread::sleep, and explains the concrete consequence: blocking stalls the worker thread 'and every other task queued to that same worker, not just yours.' It also routes blocking work through tokio::task::spawn_blocking. No fail criteria hold."
1337
+ },
1338
+ "passed": true
1339
+ },
1340
+ {
1341
+ "output": "The two things you named are exactly the two traps this skill flags:\n\n**1. Reading the config file — do not use `std::fs::read_to_string` inside the `async fn`.**\n\nA synchronous file read on a multi-threaded tokio runtime blocks the worker thread it's running on. Every other task scheduled on that worker stalls for the duration of the read, not just yours. Use the async equivalent:\n\n```rust\n// Prefer this (tokio's async file I/O)\nlet contents = tokio::fs::read_to_string(path).await?;\n```\n\n`tokio::fs` internally routes the real blocking syscall through a blocking thread pool, so the async worker is never held. If for some reason you must call a synchronous, genuinely blocking function on the config path, wrap it explicitly:\n\n```rust\nlet contents = tokio::task::spawn_blocking(move || {\n std::fs::read_to_string(&path)\n}).await??; // outer ? for JoinError, inner ? for io::Error\n```\n\nThat `spawn_blocking` is also the right home for any CPU-bound parse/validation step on the config that would otherwise hog the worker.\n\n**2. Sleeping between retries — do not use `std::thread::sleep`.**\n\nSame problem: it blocks the worker thread even though it looks like \"just waiting.\" Use tokio's timer:\n\n```rust\ntokio::time::sleep(Duration::from_millis(backoff_ms)).await;\n```\n\nOther things to watch on this path:\n\n- **Holding a lock across `.await`.** If your retry loop shares state (a cached config, a counter), don't hold a `std::sync::Mutex` guard across the `.await` point — that can deadlock or block the executor. Use `tokio::sync::Mutex` for state genuinely held across awaits, or drop the guard before awaiting.\n- **Error shape for the retry loop.** A library module should surface a `thiserror` enum with `#[from]`/`#[source]` on the variants wrapping `io::Error` and the parse error; a binary/entry point can use `anyhow::Result` with `.context(\"loading config\")` at the `?`. Don't `.unwrap()` the read or the parse — that input is external/untrusted and can genuinely fail.\n- **Bounding the retries.** Give the loop a max attempt count and a cap on backoff (or use a crate like `backon`/`tokio-retry`) so a persistently missing/unparseable config doesn't spin forever.\n- **Match the crate.** Check `Cargo.toml` for the declared edition/MSRV and confirm whether the project already has a preferred config-loading or retry convention in neighboring modules — mirror it rather than introducing a second style.\n\nBefore calling it done: `cargo build --all-targets`, `cargo clippy --all-targets -- -D warnings`, `cargo fmt --check`, and `cargo test` should all pass, and there should be no blocking call anywhere on this path outside `spawn_blocking`.",
1342
+ "outputSha256": "3d650c44ede8cb6c9f0795b29701b2a832790f257735a3a36b69f52fbb8cd051",
1343
+ "promptSha256": "c7f9f4faf4b75e041af6e5ed8faebd7745c4877596ed9eeffb26025abee596cb",
1344
+ "deterministic": [],
1345
+ "judge": {
1346
+ "verdict": "pass",
1347
+ "reason": "Criterion 1: names tokio::fs::read_to_string(path).await? instead of std::fs. Criterion 2: names tokio::time::sleep(Duration::from_millis(backoff_ms)).await instead of std::thread::sleep. Criterion 3: explains blocking stalls the worker thread so 'Every other task scheduled on that worker stalls for the duration of the read, not just yours.' Also routes blocking work through tokio::task::spawn_blocking. No fail criteria triggered."
1348
+ },
1349
+ "passed": true
1350
+ }
1351
+ ]
1352
+ }
1353
+ ],
1354
+ "verdict": "fail",
1355
+ "scope": "bundled",
1356
+ "skillDigest": "22dfae1e78e1fceacc311fc8f50991305f806ddb4804edd51cdb2efd4c3794b3",
1357
+ "catalogDigest": "58f55f0866f2076d0959c6ddb3326a6ed904e5fb06de729cf465181d86b85e90",
1358
+ "judgePromptVersion": "2026-09-25.1",
1359
+ "runner": "deepseek",
1360
+ "model": "deepseek-chat",
1361
+ "runnerPromptVersion": "2026-09-25.1",
1362
+ "recordedAt": "2026-09-25T22:11:49.462Z",
1363
+ "judge": "deepseek",
1364
+ "judgeModel": "deepseek-chat"
1365
+ },
1366
+ {
1367
+ "schemaVersion": "1.0.0",
1368
+ "skillId": "rust/rust-testing",
1369
+ "strictness": "high",
1370
+ "trials": 10,
1371
+ "triggerAccuracy": {
1372
+ "truePositive": 3,
1373
+ "falsePositive": 1,
1374
+ "positives": 7,
1375
+ "negatives": 7
1376
+ },
1377
+ "evidence": "authored",
1378
+ "scenarios": [
1379
+ {
1380
+ "id": "trigger-positive-1",
1381
+ "kind": "trigger-positive",
1382
+ "prompt": "I just wrote parse_header and it has zero coverage -- can you set it up with tests the idiomatic way, inline in the same file?",
1383
+ "strictness": "high",
1384
+ "trials": 1,
1385
+ "passes": 0,
1386
+ "passRate": 0,
1387
+ "passAtK": 0,
1388
+ "grader": "trigger-rank-fork-family",
1389
+ "status": "ran",
1390
+ "deterministic": true
1391
+ },
1392
+ {
1393
+ "id": "trigger-positive-2",
1394
+ "kind": "trigger-positive",
1395
+ "prompt": "Add table-style test cases for this Rust validator with a Vec of cases",
1396
+ "strictness": "high",
1397
+ "trials": 1,
1398
+ "passes": 1,
1399
+ "passRate": 1,
1400
+ "passAtK": 1,
1401
+ "grader": "trigger-rank-fork-family",
1402
+ "status": "ran",
1403
+ "deterministic": true
1404
+ },
1405
+ {
1406
+ "id": "trigger-positive-3",
1407
+ "kind": "trigger-positive",
1408
+ "prompt": "`cargo test -p order` is red on main after my last change -- test_reject_duplicate_line_item is failing, can you dig into why and fix it?",
1409
+ "strictness": "high",
1410
+ "trials": 1,
1411
+ "passes": 0,
1412
+ "passRate": 0,
1413
+ "passAtK": 0,
1414
+ "grader": "trigger-rank-fork-family",
1415
+ "status": "ran",
1416
+ "deterministic": true
1417
+ },
1418
+ {
1419
+ "id": "trigger-positive-4",
1420
+ "kind": "trigger-positive",
1421
+ "prompt": "Add a proptest for this Rust parser function",
1422
+ "strictness": "high",
1423
+ "trials": 1,
1424
+ "passes": 1,
1425
+ "passRate": 1,
1426
+ "passAtK": 1,
1427
+ "grader": "trigger-rank-fork-family",
1428
+ "status": "ran",
1429
+ "deterministic": true
1430
+ },
1431
+ {
1432
+ "id": "trigger-positive-5",
1433
+ "kind": "trigger-positive",
1434
+ "prompt": "I want hard numbers before I optimize parse_batch -- what's the right way to measure it so the compiler doesn't just optimize the whole thing away?",
1435
+ "strictness": "high",
1436
+ "trials": 1,
1437
+ "passes": 0,
1438
+ "passRate": 0,
1439
+ "passAtK": 0,
1440
+ "grader": "trigger-rank-fork-family",
1441
+ "status": "ran",
1442
+ "deterministic": true
1443
+ },
1444
+ {
1445
+ "id": "trigger-positive-6",
1446
+ "kind": "trigger-positive",
1447
+ "prompt": "Test this async fn with #[tokio::test] and join the spawned task properly",
1448
+ "strictness": "high",
1449
+ "trials": 1,
1450
+ "passes": 1,
1451
+ "passRate": 1,
1452
+ "passAtK": 1,
1453
+ "grader": "trigger-rank-fork-family",
1454
+ "status": "ran",
1455
+ "deterministic": true
1456
+ },
1457
+ {
1458
+ "id": "trigger-positive-7",
1459
+ "kind": "trigger-positive",
1460
+ "prompt": "Right now only the internals of this crate have coverage -- how do I set up tests that exercise it the same way an external consumer would, calling only what's exported?",
1461
+ "strictness": "high",
1462
+ "trials": 1,
1463
+ "passes": 0,
1464
+ "passRate": 0,
1465
+ "passAtK": 0,
1466
+ "grader": "trigger-rank-fork-family",
1467
+ "status": "ran",
1468
+ "deterministic": true
1469
+ },
1470
+ {
1471
+ "id": "trigger-negative-1",
1472
+ "kind": "trigger-negative",
1473
+ "prompt": "Write pytest tests for this Python function",
1474
+ "strictness": "high",
1475
+ "trials": 1,
1476
+ "passes": 1,
1477
+ "passRate": 1,
1478
+ "passAtK": 1,
1479
+ "grader": "trigger-rank-fork-family",
1480
+ "status": "ran",
1481
+ "deterministic": true
1482
+ },
1483
+ {
1484
+ "id": "trigger-negative-2",
1485
+ "kind": "trigger-negative",
1486
+ "prompt": "Add Jest tests for this React component",
1487
+ "strictness": "high",
1488
+ "trials": 1,
1489
+ "passes": 1,
1490
+ "passRate": 1,
1491
+ "passAtK": 1,
1492
+ "grader": "trigger-rank-fork-family",
1493
+ "status": "ran",
1494
+ "deterministic": true
1495
+ },
1496
+ {
1497
+ "id": "trigger-negative-3",
1498
+ "kind": "trigger-negative",
1499
+ "prompt": "Review this Rust diff for unwrap panics and unsafe blocks",
1500
+ "strictness": "high",
1501
+ "trials": 1,
1502
+ "passes": 1,
1503
+ "passRate": 1,
1504
+ "passAtK": 1,
1505
+ "grader": "trigger-rank-fork-family",
1506
+ "status": "ran",
1507
+ "deterministic": true
1508
+ },
1509
+ {
1510
+ "id": "trigger-negative-4",
1511
+ "kind": "trigger-negative",
1512
+ "prompt": "Fix the cargo build error in this crate",
1513
+ "strictness": "high",
1514
+ "trials": 1,
1515
+ "passes": 0,
1516
+ "passRate": 0,
1517
+ "passAtK": 0,
1518
+ "grader": "trigger-rank-fork-family",
1519
+ "status": "ran",
1520
+ "deterministic": true
1521
+ },
1522
+ {
1523
+ "id": "trigger-negative-5",
1524
+ "kind": "trigger-negative",
1525
+ "prompt": "Implement a new Rust service that calls this downstream API",
1526
+ "strictness": "high",
1527
+ "trials": 1,
1528
+ "passes": 1,
1529
+ "passRate": 1,
1530
+ "passAtK": 1,
1531
+ "grader": "trigger-rank-fork-family",
1532
+ "status": "ran",
1533
+ "deterministic": true
1534
+ },
1535
+ {
1536
+ "id": "trigger-negative-6",
1537
+ "kind": "trigger-negative",
1538
+ "prompt": "Write Go table-driven tests for this function",
1539
+ "strictness": "high",
1540
+ "trials": 1,
1541
+ "passes": 1,
1542
+ "passRate": 1,
1543
+ "passAtK": 1,
1544
+ "grader": "trigger-rank-fork-family",
1545
+ "status": "ran",
1546
+ "deterministic": true
1547
+ },
1548
+ {
1549
+ "id": "trigger-negative-7",
1550
+ "kind": "trigger-negative",
1551
+ "prompt": "Add unit tests in a Go _test.go file for this package",
1552
+ "strictness": "high",
1553
+ "trials": 1,
1554
+ "passes": 1,
1555
+ "passRate": 1,
1556
+ "passAtK": 1,
1557
+ "grader": "trigger-rank-fork-family",
1558
+ "status": "ran",
1559
+ "deterministic": true
1560
+ },
1561
+ {
1562
+ "id": "error-variant-assertion",
1563
+ "kind": "behavior",
1564
+ "prompt": "I need to add several test cases for this Rust function that validates an order and returns a Result. How should I structure them and check the error cases?",
1565
+ "strictness": "high",
1566
+ "trials": 10,
1567
+ "passes": 10,
1568
+ "passRate": 1,
1569
+ "passAtK": 1,
1570
+ "grader": "judge",
1571
+ "status": "ran",
1572
+ "trialRecords": [
1573
+ {
1574
+ "output": "For a function that validates an order and returns a `Result`, structure the cases two ways depending on whether they share an assertion shape:\n\n**Happy path + error variants as separate tests when each asserts something different**, or **a table of cases when they share the same shape**. The common, idiomatic pattern for several cases that all assert \"input → expected outcome\":\n\n```rust\n#[cfg(test)]\nmod tests {\n use super::*;\n\n struct Case {\n name: &'static str,\n input: Order,\n expected: Result<(), OrderError>,\n }\n\n #[test]\n fn validate_order_cases() {\n let cases = vec![\n Case {\n name: \"valid order\",\n input: Order { id: 1, items: vec![item(1)], total: 100 },\n expected: Ok(()),\n },\n Case {\n name: \"empty items\",\n input: Order { id: 2, items: vec![], total: 100 },\n expected: Err(OrderError::NoItems),\n },\n Case {\n name: \"zero total\",\n input: Order { id: 3, items: vec![item(1)], total: 0 },\n expected: Err(OrderError::InvalidTotal { total: 0 }),\n },\n Case {\n name: \"negative total\",\n input: Order { id: 4, items: vec![item(1)], total: -5 },\n expected: Err(OrderError::InvalidTotal { total: -5 }),\n },\n ];\n\n for case in cases {\n assert_eq!(\n validate_order(&case.input),\n case.expected,\n \"case: {}\",\n case.name,\n );\n }\n }\n}\n```\n\nThis requires `OrderError: PartialEq` (usually derivable). The `\"case: {}\"` message is what makes a table test debuggable — without it, a failure only tells you the position in the `Vec`.\n\n**On checking error cases specifically:** assert the *specific* variant, not just that it errored. `assert!(result.is_err())` passes for `NoItems` when you expected `InvalidTotal`, so it hides regressions. Two ways to pin the variant:\n\n- `assert_eq!` against the full `Err(OrderError::InvalidTotal { total: 0 })` when the error derives `PartialEq` and you want to check the payload too.\n- `matches!` when the error isn't `PartialEq`, or you only care about the variant, not every field:\n\n```rust\nlet err = validate_order(&order).unwrap_err();\nassert!(\n matches!(err, OrderError::InvalidTotal { .. }),\n \"expected InvalidTotal, got {err:?}\",\n);\n```\n\nIf the error has fields with values you care about, match and destructure:\n\n```rust\nlet err = validate_order(&order).unwrap_err();\nmatch err {\n OrderError::InvalidTotal { total } => assert_eq!(total, -5),\n other => panic!(\"expected InvalidTotal, got {other:?}\"),\n}\n```\n\n**Plan the cases around the function's actual decision points**, not just happy/error:\n\n- happy path (a fully valid order)\n- each distinct validation branch: empty items, zero/negative total, missing id, duplicate items — one case per rule the function enforces\n- boundary values: total exactly at a min/max, exactly the item-count limit, empty vs one item\n- error cases as above, each asserting its own variant\n\n**Other conventions to match, per the skill:**\n\n- If the crate uses a `tests/` integration directory for this kind of test, put it there instead of a `#[cfg(test)]` module; read a neighboring test first and match its style, fixture location (`tests/fixtures/` or `testdata/`), and whether it already uses `pretty_assertions`.\n- Only test files change — don't touch the source under test.\n- If any case reveals a real bug in `validate_order`, report it rather than editing the source to make the test pass.\n\n**Run `cargo test` and report** with the count:\n\n```\nGenerated: src/order/validate.rs (tests module)\n - 4 table-style cases + 3 variant-specific error tests, all passing under cargo test\n```",
1575
+ "outputSha256": "809334ebf6905225356e89ac198ebe045483fe986a0cc085f90a59c640df8b89",
1576
+ "promptSha256": "c107ab33550812464aa3b85a7a72ac299878517f6f1070b6d77896bd5a11bcb1",
1577
+ "deterministic": [],
1578
+ "judge": {
1579
+ "verdict": "pass",
1580
+ "reason": "Answer shows a Vec<Case> table with input/expected fields iterated in one test (validate_order_cases), asserts specific variants via assert_eq!(..., Err(OrderError::InvalidTotal{total:0})) and matches!(err, OrderError::InvalidTotal{..}), and explains why: 'assert!(result.is_err()) passes for NoItems when you expected InvalidTotal, so it hides regressions.'"
1581
+ },
1582
+ "passed": true
1583
+ },
1584
+ {
1585
+ "output": "Structure them as a table-driven test and check error cases by asserting the exact variant, not just that it failed.\n\n**Layout** — put a `#[cfg(test)] mod tests` block in the same file as the function, mirroring whatever the crate already does. One `#[test]` function per behavior, named for what it verifies (`rejects_negative_quantity`, not `test_validation_2`).\n\n**Table shape** — since every case is \"input → expected outcome,\" a local case struct iterated in one test beats a dozen near-duplicate functions:\n\n```rust\n#[cfg(test)]\nmod tests {\n use super::*;\n\n struct Case {\n name: &'static str,\n order: Order,\n expected: Result<(), OrderError>,\n }\n\n #[test]\n fn validate_order_cases() {\n let cases = vec![\n Case {\n name: \"valid order\",\n order: Order { quantity: 1, ..Default::default() },\n expected: Ok(()),\n },\n Case {\n name: \"rejects zero quantity\",\n order: Order { quantity: 0, ..Default::default() },\n expected: Err(OrderError::InvalidQuantity),\n },\n Case {\n name: \"rejects missing customer\",\n order: Order { customer_id: None, ..Default::default() },\n expected: Err(OrderError::MissingCustomer),\n },\n ];\n\n for case in cases {\n assert_eq!(\n validate_order(&case.order),\n case.expected,\n \"case: {}\",\n case.name,\n );\n }\n }\n}\n```\n\n**Error cases** — the key point from the rules: don't write `assert!(result.is_err())`. That passes whether you got `MissingCustomer` or a completely unrelated failure, so the test stops catching regressions. Two options:\n\n- If `OrderError` derives `PartialEq` (`#[derive(Debug, PartialEq)]`), `assert_eq!(result, Err(OrderError::InvalidQuantity))` above is cleanest — and the `Debug` bound gives a good failure message.\n- If it doesn't implement `PartialEq` (e.g. it wraps a non-comparable field), match on the variant:\n\n```rust\nlet err = validate_order(&order).unwrap_err();\nassert!(matches!(err, OrderError::InvalidQuantity), \"got {err:?}\");\n```\n\nBoth pin down the specific variant. Split the `Ok` and `Err` assertions if the case struct can't hold a `Result` because `PartialEq` isn't derived anyway — a small `Outcome` enum or a `matches!` on the whole result works too.\n\nTwo things I'd avoid: don't synchronize anything with `sleep` (irrelevant here since validation is sync, but it's the usual trap if this later grows an async path), and don't touch `validate_order` itself to make a case pass — if a test catches a real bug, report it rather than editing the function.\n\nRun `cargo test` after; fix failures in the test, up to a few rounds. Then confirm `git status` shows only the test module touched.",
1586
+ "outputSha256": "b09ac9c3af961bd6a12f2a3c2e720ee92f111d160dad9be855aeabb510b38ffe",
1587
+ "promptSha256": "c107ab33550812464aa3b85a7a72ac299878517f6f1070b6d77896bd5a11bcb1",
1588
+ "deterministic": [],
1589
+ "judge": {
1590
+ "verdict": "pass",
1591
+ "reason": "Answer shows a Vec<Case> table with name/order/expected fields iterated in one test (validate_order_cases), asserts specific variants via assert_eq!(result, Err(OrderError::InvalidQuantity)) and matches!(err, OrderError::InvalidQuantity), and explains why: is_err() passes for unrelated failures so the test stops catching regressions."
1592
+ },
1593
+ "passed": true
1594
+ },
1595
+ {
1596
+ "output": "For a `Result`-returning validator, structure the cases so each one names the exact failure it proves — here's the shape I'd use.\n\n**1. Find the crate's existing convention first.** Check whether neighboring tests use a `#[cfg(test)] mod tests` block beside the code or a `tests/` integration file, and whether cases are written as one-off `#[test]` functions or as a table. Match that. If nothing exists yet, the table style below is the default I'd pick.\n\n**2. Table style for validator cases.** A validator naturally has one input shape and one assertion shape (expected `Ok(..)` or expected error variant), which is exactly when a table beats near-duplicate test functions:\n\n```rust\n#[cfg(test)]\nmod tests {\n use super::*;\n\n struct Case {\n name: &'static str,\n order: Order,\n expected: Result<(), OrderError>,\n }\n\n #[test]\n fn validates_orders() {\n let cases = vec![\n Case {\n name: \"valid order\",\n order: Order { /* happy path fields */ },\n expected: Ok(()),\n },\n Case {\n name: \"empty line items\",\n order: Order { items: vec![], ..valid() },\n expected: Err(OrderError::EmptyOrder),\n },\n // ... boundary/error cases\n ];\n\n for case in cases {\n let result = validate_order(&case.order);\n assert_eq!(result, case.expected, \"case: {}\", case.name);\n }\n }\n}\n```\n\nThis only works if `OrderError` and whatever sits inside `Ok` implement `PartialEq` (+ `Debug` for the assert message). If they don't and you can't add derives (you shouldn't be editing source), fall back to per-case `matches!` assertions instead of `assert_eq!`.\n\n**3. Error cases — assert the specific variant, never just `is_err()`.** This is the rule that matters most here:\n\n```rust\n// No — passes for the wrong failure too\nassert!(validate_order(&order).is_err());\n\n// Yes — pins down which failure\nassert_eq!(validate_order(&order), Err(OrderError::EmptyOrder));\n\n// Or, when the error carries data or lacks PartialEq\nlet err = validate_order(&order).unwrap_err();\nassert!(matches!(err, OrderError::UnknownProduct { ref sku, .. } if sku == \"abc\"));\n```\n\nAn `is_err()`-only check happily passes if the validator starts failing for an unrelated reason, so it can't catch a regression. Assert the variant.\n\n**4. Cover these case categories explicitly:**\n- Happy path (valid order → `Ok`)\n- Boundary values: zero/empty line items, zero quantity, zero or negative totals, max-length fields\n- Each distinct error variant the function can return — one case per variant, including the branches that carry payload (bad sku, unknown product)\n- Any \"first error wins\" ordering if the validator short-circuits, so you lock in which error takes precedence\n\n**5. Add `#[test]` naming and fixture data per convention.** Name each test for the behavior, not `test1`. If cases need large golden inputs, put them under the project's existing fixture path (`tests/fixtures/` or `testdata/`), not inline.\n\n**6. Run `cargo test` and iterate (max 3 rounds).** If a case fails, fix the test unless it has genuinely caught a real bug — in that case report the bug rather than silently editing the source; this workflow only touches test files.\n\nConstraints to keep in mind throughout: don't modify source under test, don't reach for `Order`/`OrderError` internals that aren't public, and don't loosen a failing assertion to `is_err()`. `git status` at the end should show only test files (and fixtures) changed.",
1597
+ "outputSha256": "a9529a9c98de5cb4a909905987800d8c7fb2c65620218de1da2f22c4fff4eb74",
1598
+ "promptSha256": "c107ab33550812464aa3b85a7a72ac299878517f6f1070b6d77896bd5a11bcb1",
1599
+ "deterministic": [],
1600
+ "judge": {
1601
+ "verdict": "pass",
1602
+ "reason": "Answer shows a Vec<Case> table with name/order/expected fields iterated in one test (section 2), asserts the specific variant via assert_eq!(..., Err(OrderError::EmptyOrder)) and matches!(err, OrderError::UnknownProduct{..}) (section 3), and explains why: is_err() passes for wrong failures and can't catch regressions. Satisfies all criteria."
1603
+ },
1604
+ "passed": true
1605
+ },
1606
+ {
1607
+ "output": "Structure them as a table-style test and assert the exact error variant, not just `is_err()`.\n\n**Table-style shape**\n\n```rust\n#[cfg(test)]\nmod tests {\n use super::*;\n\n struct Case {\n name: &'static str,\n input: Order,\n expected: Result<ValidatedOrder, OrderError>,\n }\n\n #[test]\n fn validates_orders() {\n let cases = vec![\n Case {\n name: \"valid order\",\n input: Order { id: 1, items: vec![/* ... */] },\n expected: Ok(ValidatedOrder { /* ... */ }),\n },\n Case {\n name: \"empty items rejected\",\n input: Order { id: 2, items: vec![] },\n expected: Err(OrderError::EmptyCart),\n },\n Case {\n name: \"unknown item rejected\",\n input: Order { id: 3, items: vec![/* unknown sku */] },\n expected: Err(OrderError::UnknownItem { sku: \"XYZ\".into() }),\n },\n ];\n\n for case in cases {\n let got = validate_order(&case.input);\n assert_eq!(got, case.expected, \"case: {}\", case.name);\n }\n }\n}\n```\n\nThat only works if `OrderError` and the success type implement `PartialEq` + `Debug`. If they don't, assert the exact variant with `matches!` instead of loosening to `is_err()`:\n\n```rust\nlet err = validate_order(&case.input).unwrap_err();\nassert!(\n matches!(err, OrderError::UnknownItem { ref sku } if sku == \"XYZ\"),\n \"case {}: expected UnknownItem, got {:?}\",\n case.name,\n err\n);\n```\n\n**Which cases to cover**\n\n- Happy path: a fully valid order.\n- Boundary/edge inputs: empty items, zero quantity, exactly at a quantity/amount limit, at limit minus one.\n- Each distinct error path your function can return — one case per `OrderError` variant, so a new variant added later has a visible gap until a case is added.\n- Precedence, if the function checks conditions in order: an input that violates two rules at once, asserting the one it documents as returned, so reordering checks is caught.\n\n**Two things to avoid**\n\n- `assert!(result.is_err())` on its own — it passes just as happily for the wrong error, hiding regressions.\n- Many near-duplicate `#[test] fn` functions per case; hoist the shared assertion shape into the loop above and name each case via the `name` field, so a failing case reports which one broke.\n\nMatch whatever convention the crate already uses (a `Vec<Case>` vs a `rstest`-style parametrization vs a `datatest` fixture) before introducing this shape, and keep the test in the `#[cfg(test)] mod tests` beside the function unless the crate already puts this kind of check under `tests/`.",
1608
+ "outputSha256": "e1a709108a1b072e2b27c3df0d6d3315e4f7a1d16a7a461618eec3a230b29bf3",
1609
+ "promptSha256": "c107ab33550812464aa3b85a7a72ac299878517f6f1070b6d77896bd5a11bcb1",
1610
+ "deterministic": [],
1611
+ "judge": {
1612
+ "verdict": "pass",
1613
+ "reason": "Answer shows a Vec<Case> table with input/expected fields iterated in one test (validates_orders loop), and concrete variant assertions: assert_eq!(got, case.expected) with Err(OrderError::EmptyCart)/UnknownItem, plus matches!(err, OrderError::UnknownItem { ref sku } if sku == \"XYZ\"). It explains why: is_err() 'passes just as happily for the wrong error, hiding regressions.'"
1614
+ },
1615
+ "passed": true
1616
+ },
1617
+ {
1618
+ "output": "Without seeing the function, I can give you the shape that works well for this and matches how Rust test suites are usually organized — you'll want to adjust names/types to your actual function.\n\n**Structure: one table-style test plus a few focused ones**\n\nIf your validation function has the same assertion shape across many inputs (input → expected outcome), put the cases in a `Vec` of a local struct and drive them from a single `#[test]`. That avoids a wall of near-duplicate functions.\n\n```rust\n#[cfg(test)]\nmod tests {\n use super::*;\n\n struct Case {\n name: &'static str,\n input: Order,\n expected: Result<(), OrderError>,\n }\n\n #[test]\n fn validates_orders() {\n let cases = vec![\n Case {\n name: \"valid order\",\n input: Order { /* ... */ },\n expected: Ok(()),\n },\n Case {\n name: \"empty line items\",\n input: Order { items: vec![], /* ... */ },\n expected: Err(OrderError::EmptyOrder),\n },\n Case {\n name: \"negative quantity\",\n input: Order { /* quantity: -1 */ },\n expected: Err(OrderError::InvalidQuantity { quantity: -1 }),\n },\n // ... more cases\n ];\n\n for case in cases {\n let result = validate_order(&case.input);\n assert_eq!(result, case.expected, \"case: {}\", case.name);\n }\n }\n}\n```\n\nTwo things make this work:\n\n1. **The case struct's `expected` field is the `Result` itself**, not a bool. That forces you to name the exact error you expect, and the failure message (`case: <name>`) tells you which row broke.\n2. **`assert_eq!` on the `Result` requires `OrderError: PartialEq + Debug`.** If your error type doesn't derive those, either add the derives (that's a source change — see the caveat below) or fall back to matching.\n\n**Checking error cases specifically**\n\nThe key rule: don't assert `result.is_err()`. That passes just as happily for the wrong error as for the right one, so it silently stops catching regressions. Assert the exact variant.\n\nIf `OrderError` implements `PartialEq`, `assert_eq!` above is enough. If it doesn't, match on the variant:\n\n```rust\nlet err = validate_order(&case.input).unwrap_err();\nassert!(\n matches!(err, OrderError::InvalidQuantity { quantity: -1 }),\n \"unexpected error: {err:?}\"\n);\n\n// or with a bound field, capture and check it:\nmatch validate_order(&case.input) {\n Err(OrderError::MissingField { field }) => assert_eq!(field, \"customer_id\"),\n other => panic!(\"expected MissingField, got {other:?}\"),\n}\n```\n\nThe `match` with an explicit `other => panic!(...)` arm is the most robust — it fails loudly with the actual value if the wrong variant comes back.\n\n**If you need the cases as separate functions instead of a table**\n\nThat's fine when the assertion shape genuinely differs per case (some check a side effect, some check a returned value). Just name each one for the behavior, not `test1`/`test2`:\n\n```rust\n#[test] fn rejects_empty_order() { /* ... */ }\n#[test] fn rejects_negative_quantity() { /* ... */ }\n#[test] fn accepts_minimal_valid_order() { /* ... */ }\n```\n\n**Edge cases worth covering for a validator**\n\n- The happy path with a fully populated order.\n- Zero/empty/boundary: empty item list, quantity of `0`, quantity at `u32::MAX`-style limits, an order with exactly the minimum allowed items.\n- Each distinct error variant you return — one case per variant, so a removed or renamed variant surfaces as a compile/test failure.\n- Whitespace/formatting on any string fields if they're validated.\n\n**One caveat about changing the error type**\n\nAdding `#[derive(PartialEq)]` to `OrderError` so `assert_eq!` works is a change to source, not a test file. If the error type already derives `PartialEq + Debug`, use the table form as-is. If it doesn't and you'd rather not touch source, stick with the `matches!`/`match` approach — that needs only `Debug`, which error types almost always have.\n\nMatch the layout your crate already uses: if it has a `tests/` directory for this kind of thing, the table goes there as an integration test; if neighbors use `#[cfg(test)] mod tests` beside the code, do that. And run it with `cargo test` — every row should go green before you call it done.",
1619
+ "outputSha256": "544b3514e2b5aaae81e40c57f1333797938b07c9806a0718d438d5c5fa65f5ec",
1620
+ "promptSha256": "c107ab33550812464aa3b85a7a72ac299878517f6f1070b6d77896bd5a11bcb1",
1621
+ "deterministic": [],
1622
+ "judge": {
1623
+ "verdict": "pass",
1624
+ "reason": "Answer shows a Vec<Case> table with input/expected fields iterated in one test (validates_orders), concrete variant assertions (assert_eq! on Result, matches!(err, OrderError::InvalidQuantity{...}), match with panic arm), and explains why: 'don't assert result.is_err()... passes just as happily for the wrong error... silently stops catching regressions.'"
1625
+ },
1626
+ "passed": true
1627
+ },
1628
+ {
1629
+ "output": "Without seeing the function itself, the shape I'd recommend (matching the conventions this skill's Step 1 would have you read from the crate first) is:\n\n**Structural choice: table-style, not one test per case.**\n\nIf the cases share the same assertion shape — same call, same expected outcome kind — collect them into a `Vec` of a local case struct and iterate in one `#[test]`. Near-duplicate test functions for each invalid input rot quickly and make it hard to see the covered case space at a glance.\n\n```rust\n#[cfg(test)]\nmod tests {\n use super::*;\n\n struct Case {\n name: &'static str,\n input: Order,\n expected: Result<Validated, OrderError>,\n }\n\n #[test]\n fn validate_order_cases() {\n let cases = vec![\n Case {\n name: \"valid order\",\n input: sample_order(),\n expected: Ok(Validated { /* ... */ }),\n },\n Case {\n name: \"empty line items rejected\",\n input: Order { items: vec![], ..sample_order() },\n expected: Err(OrderError::EmptyOrder),\n },\n Case {\n name: \"zero quantity rejected\",\n input: order_with_item(Item { qty: 0, ..default_item() }),\n expected: Err(OrderError::InvalidQuantity { qty: 0 }),\n },\n ];\n\n for case in cases {\n let got = validate_order(&case.input);\n assert_eq!(got, case.expected, \"case: {}\", case.name);\n }\n }\n}\n```\n\nThe `assert_eq!` there assumes both `Validated` and `OrderError` implement `PartialEq` and `Debug` — if they don't, add `#[derive(PartialEq, Debug)]` or fall back to the pattern-match form below. The `\"case: {}\"` message matters: without it a table failure tells you *something* mismatched but not which row.\n\n**Error cases: assert the specific variant, never `is_err()`.**\n\n`assert!(validate_order(&bad).is_err())` passes just as happily if the function returns the *wrong* error — it only tells you some failure happened, not that the right one did. Two ways to pin it down:\n\n```rust\n// (a) exact match, when the error derives PartialEq + Debug\nassert_eq!(\n validate_order(&bad),\n Err(OrderError::InvalidQuantity { qty: 0 })\n);\n\n// (b) structural match, when it doesn't (e.g. wraps a Box<dyn Error> or an inner type)\nmatch validate_order(&bad) {\n Err(OrderError::InvalidQuantity { qty }) => assert_eq!(qty, 0),\n other => panic!(\"expected InvalidQuantity, got {other:?}\"),\n}\n```\n\nForm (b) is the one to reach for when the error carries a source/boxed cause you can't `PartialEq` cleanly — it still asserts the *variant and its fields* rather than just \"an error\".\n\n**Coverage to aim for on a validator:** one happy path, one per rejection rule (empty, zero/negative quantity, missing required field, boundary values on both sides of any limit), and — if the error type has more than one variant — one case per reachable variant, so the table itself documents the full rejection surface.\n\n**Two things to check before writing any of it**, both from Step 1:\n\n- **Does the crate already have a `Result`-comparison helper or a table idiom?** If neighboring modules use `pretty_assertions` or a `assert_err!` macro, use theirs; don't introduce a second style into the same crate.\n- **Is `validate_order` async?** If so the test needs `#[tokio::test]` and, if it spawns anything, a join on the `JoinHandle` before asserting — no `tokio::time::sleep` to \"let it finish.\"\n\nFinally: if `Order` or `Validated` is awkward to build inline (many fields, invariants), a `sample_order()`/`default_item()` fixture helper at the top of the test module keeps each `Case` to just the field it's actually varying — which is the point of the table. I'd want to see `validate_order`'s signature and its error enum before finalizing the exact variant names and whether form (a) or (b) applies.",
1630
+ "outputSha256": "f88217056821dc3a4912574db6382a34bab10bce4991e46e1764bc66da28bc86",
1631
+ "promptSha256": "c107ab33550812464aa3b85a7a72ac299878517f6f1070b6d77896bd5a11bcb1",
1632
+ "deterministic": [],
1633
+ "judge": {
1634
+ "verdict": "pass",
1635
+ "reason": "Answer uses a Vec<Case> table with input/expected fields iterated in one #[test] (criterion 1). It shows concrete variant assertions: assert_eq!(validate_order(&bad), Err(OrderError::InvalidQuantity { qty: 0 })) and a match arm Err(OrderError::InvalidQuantity { qty }) => assert_eq!(qty, 0) (criterion 2). It explains is_err() passes even if the wrong error is returned, verifying which validation failed (criterion 3)."
1636
+ },
1637
+ "passed": true
1638
+ },
1639
+ {
1640
+ "output": "Structure the cases as one table-driven `#[test]` (a `Vec` of a small local case struct) rather than a pile of near-identical test functions, and assert the *specific* error variant in each error case — not just `result.is_err()`.\n\nSketch, matching a typical crate layout (extend/replace your crate's own names, error type, and module path):\n\n```rust\n#[cfg(test)]\nmod tests {\n use super::*;\n use crate::OrderError;\n\n struct Case {\n name: &'static str,\n order: Order,\n expected: Result<ValidatedOrder, OrderError>,\n }\n\n #[test]\n fn validate_order_cases() {\n let cases = vec![\n Case {\n name: \"happy path: well-formed order\",\n order: Order { /* ... */ },\n expected: Ok(ValidatedOrder { /* ... */ }),\n },\n Case {\n name: \"empty line items is rejected\",\n order: Order { items: vec![], ..base() },\n expected: Err(OrderError::NoItems),\n },\n Case {\n name: \"zero-quantity item is rejected\",\n order: Order { items: vec![Item { qty: 0, .. }], ..base() },\n expected: Err(OrderError::InvalidQuantity { index: 0, qty: 0 }),\n },\n Case {\n name: \"total exceeding limit is rejected\",\n order: Order { total_cents: LIMIT + 1, ..base() },\n expected: Err(OrderError::TotalExceedsLimit { total: LIMIT + 1, limit: LIMIT }),\n },\n ];\n\n for case in cases {\n let got = validate_order(&case.order);\n assert_eq!(\n got, case.expected,\n \"case `{}` failed: got {:?}, expected {:?}\",\n case.name, got, case.expected\n );\n }\n }\n}\n```\n\nKey points:\n\n- **Table shape**: a local `Case` struct named by `name`, holding input + expected result. One loop, one assertion site — failures report the case name so you don't get a wall of anonymous assertions.\n- **Specific error variant**: compare against the exact `Err(OrderError::Variant { .. })` value. This requires `OrderError: PartialEq + Debug`; if it only derives `Debug`, use `assert!(matches!(got, Err(OrderError::Variant { .. })))` per case instead of `assert_eq!`. Either way, don't settle for `is_err()` — `is_err()` passes just as readily for the wrong failure as the right one, which hides a regression next time.\n- **Coverage**: happy path, a boundary case at the limit (equal to `LIMIT`, valid) and just past it (rejected), empty/zero inputs, and any multi-field error where the index/qty in the variant matters.\n- **PartialEq availability**: if `ValidatedOrder` can't sensibly derive `PartialEq`, split the table so success cases assert selected fields and error cases stay exact on the variant.\n\nIf the crate already has a `tests/` integration directory and this function is public API, put the equivalent table there instead; keep the `#[cfg(test)]` variant only for private internals. And whatever you add must leave `cargo test` green without touching the source under test — if a case legitimately exposes a real bug, report it rather than loosening the assertion in place.",
1641
+ "outputSha256": "409c7915e3a65eeb9c50f9a0e1243cd9bc9e8106bc059c18db82af769ea170d4",
1642
+ "promptSha256": "c107ab33550812464aa3b85a7a72ac299878517f6f1070b6d77896bd5a11bcb1",
1643
+ "deterministic": [],
1644
+ "judge": {
1645
+ "verdict": "pass",
1646
+ "reason": "Answer shows a Vec<Case> table with name/order/expected fields iterated in one test (criterion 1). It asserts exact variants via assert_eq!(got, case.expected) with Err(OrderError::NoItems), InvalidQuantity{index,qty}, TotalExceedsLimit{total,limit}, and offers matches! fallback (criterion 2). It explains is_err() passes for the wrong failure and hides regressions (criterion 3)."
1647
+ },
1648
+ "passed": true
1649
+ },
1650
+ {
1651
+ "output": "Structure the cases as a table and assert the exact error variant on each failure. Concretely:\n\n**Shape:** one `#[cfg(test)] mod tests` (or extend the existing one) with a single `#[test]` that iterates a slice of a local case struct — not one test function per case — when every case has the same call-and-compare shape.\n\n```rust\n#[cfg(test)]\nmod tests {\n use super::*;\n\n struct Case {\n name: &'static str,\n input: Order,\n expected: Result<ValidatedOrder, ValidationError>,\n }\n\n #[test]\n fn validate_order_cases() {\n let cases = vec![\n Case { name: \"happy path\", input: valid_order(), expected: Ok(expected_validated()) },\n Case { name: \"empty items\", input: order_with_items(vec![]), expected: Err(ValidationError::NoItems) },\n Case { name: \"zero quantity\", input: order_with_qty(0), expected: Err(ValidationError::InvalidQuantity { quantity: 0 }) },\n Case { name: \"unknown sku\", input: order_with_sku(\"NOPE\"), expected: Err(ValidationError::UnknownSku(\"NOPE\".into())) },\n Case { name: \"total overflow\", input: order_at_i64_max(), expected: Err(ValidationError::TotalOverflow) },\n ];\n\n for case in cases {\n let got = validate_order(&case.input);\n assert_eq!(got, case.expected, \"case: {}\", case.name);\n }\n }\n}\n```\n\n**Key points:**\n\n1. **Name the behaviors**, as above — `\"empty items\"`, `\"zero quantity\"` — not `test1`/`test_x`. The `\"case: {}\"` message makes the failing case identifiable in the output.\n\n2. **Assert the specific error variant, never just `is_err()`.** `assert_eq!` on the whole `Result` works when `ValidationError` implements `PartialEq` and `Debug` (derive both if needed). If the error carries a non-comparable payload (an `io::Error`, anything without `PartialEq`), use a `matches!` instead:\n\n```rust\nassert!(\n matches!(got, Err(ValidationError::UnknownSku(ref s)) if s == \"NOPE\"),\n \"case: {} — got {:?}\", case.name, got\n);\n```\n\nThis is the point of the whole exercise: `is_err()` passes for the *wrong* failure just as readily as the right one, so a regression that swaps one error variant for another would sail through.\n\n3. **Cover the error cases deliberately:** missing required fields, empty collections, zero/negative quantities, boundary values (max/min, off-by-one around a limit), unknown/foreign IDs, and any arithmetic that can overflow or divide by zero. Then a happy path or two.\n\n4. **If `Order` / `ValidatedOrder` don't derive `PartialEq` or `Debug`,** add the derives in the test-reachable way the project already uses, or fall back to per-field `matches!` assertions. Don't compare by formatting strings into booleans.\n\n5. **No sleeps** anywhere in this — `validate_order` looks synchronous, but if any of it is async, join the future/task via `.await` on the `JoinHandle` rather than `tokio::time::sleep`.\n\n6. **Touch only test code:** the tests module and any fixture builders. If a case reveals a real bug in `validate_order`, report it rather than editing the function.\n\nThen run `cargo test` and confirm every case passes with `git status` showing only the test file changed.",
1652
+ "outputSha256": "52e86b4d3f327b375e6979bf685181a7a89c372d88a084d335f4fcd1b4a7f99b",
1653
+ "promptSha256": "c107ab33550812464aa3b85a7a72ac299878517f6f1070b6d77896bd5a11bcb1",
1654
+ "deterministic": [],
1655
+ "judge": {
1656
+ "verdict": "pass",
1657
+ "reason": "Answer shows a Vec<Case> table with name/input/expected fields iterated in one #[test] (criterion 1). It asserts specific variants via assert_eq!(got, case.expected) with Err(ValidationError::NoItems) etc., and a matches!(got, Err(ValidationError::UnknownSku(ref s)) if s == \"NOPE\") fallback (criterion 2). It explains is_err() passes for the wrong failure, so a swapped variant would sail through (criterion 3)."
1658
+ },
1659
+ "passed": true
1660
+ },
1661
+ {
1662
+ "output": "For a `Result`-returning validator, structure the tests as a table of cases in one `#[test]` function (or a small set of `#[test]` functions grouped by outcome), and assert the **specific** error variant on the failure cases — not just `is_err()`.\n\nFirst, check the crate's conventions before writing anything:\n\n1. Read `Cargo.toml` (edition, dev-dependencies) and look at 1–2 neighboring `#[cfg(test)] mod tests` blocks or `tests/` files to match the local style.\n2. See whether the error type implements `PartialEq` (then `assert_eq!` works directly) or not (then use `matches!` / a `match` with a bound field).\n\nShape it like this:\n\n```rust\n#[cfg(test)]\nmod tests {\n use super::*;\n\n struct Case {\n name: &'static str,\n input: Order,\n expected: Result<(), OrderError>,\n }\n\n #[test]\n fn validates_order() {\n let cases = vec![\n Case {\n name: \"happy path: well-formed order\",\n input: Order { /* ... */ },\n expected: Ok(()),\n },\n Case {\n name: \"empty line items\",\n input: Order { items: vec![], ..valid_order() },\n expected: Err(OrderError::EmptyOrder),\n },\n Case {\n name: \"zero quantity\",\n input: Order { items: vec![line(0)], ..valid_order() },\n expected: Err(OrderError::InvalidQuantity { sku: \"A1\".into() }),\n },\n Case {\n name: \"missing shipping address\",\n input: Order { shipping: None, ..valid_order() },\n expected: Err(OrderError::MissingShippingAddress),\n },\n ];\n\n for case in cases {\n let result = validate_order(&case.input);\n assert_eq!(result, case.expected, \"case: {}\", case.name);\n }\n }\n}\n```\n\nKey points, matching the skill's rules:\n\n- **Table of cases.** One struct holding name/input/expected, iterated in a loop, when cases share the same assertion shape. This beats a dozen near-duplicate `test_zero_qty`, `test_one_qty`, … functions, and the `name` field tells you *which* case failed.\n- **Specific error assertions.** `expected: Err(OrderError::InvalidQuantity { sku: \"A1\".into() })` pins the exact variant *and* the payload. If the error type doesn't implement `PartialEq`, swap to `assert!(matches!(result, Err(OrderError::InvalidQuantity { sku }) if sku == \"A1\"))`. Do not settle for `assert!(result.is_err())` — that passes just as happily for the wrong failure and hides regressions.\n- **Cover the three categories.** Happy path, edge cases (empty items, quantity 0, boundary counts), and each distinct error variant. If the error enum has N variants, ideally each is produced by at least one case.\n- **A fixture helper** (`valid_order()`, `line(qty)`) keeps each case focused on the one field it perturbs, via struct-update syntax.\n\nIf the input space is large/structured (many field combinations), reach for `proptest!` and assert the invariant (e.g. \"`Ok` implies every item qty > 0 and shipping is `Some`\") rather than enumerating examples by hand.\n\nTwo hard constraints from the skill:\n\n- Only test files change. If a case fails, fix the test — unless the test has genuinely caught a bug in `validate_order`, in which case report it rather than silently editing the source.\n- No `std::thread::sleep` / `tokio::time::sleep` to wait for anything; that's a flake generator. (If this validator is `async` or spawns work, use `#[tokio::test]` and join the `JoinHandle` before asserting.)\n\nThen verify: `cargo test` exits 0 with all cases passing, and `git status` shows only test/fixture files touched — no source file modified.",
1663
+ "outputSha256": "9e8c59ea135ab482cc9744216f7ce7d2ee0d7c08668fdf6cf76553f033656dc9",
1664
+ "promptSha256": "c107ab33550812464aa3b85a7a72ac299878517f6f1070b6d77896bd5a11bcb1",
1665
+ "deterministic": [],
1666
+ "judge": {
1667
+ "verdict": "pass",
1668
+ "reason": "Answer shows a Vec<Case> table with name/input/expected fields iterated in one #[test] (struct Case + for case in cases). It asserts the specific variant via assert_eq!(result, case.expected) with expected: Err(OrderError::InvalidQuantity{...}), and offers matches! fallback. It explains why: 'that passes just as happily for the wrong failure and hides regressions.'"
1669
+ },
1670
+ "passed": true
1671
+ },
1672
+ {
1673
+ "output": "Structure them as a table-style test in a `#[cfg(test)] mod tests` block (or a `tests/` file, if that's where the crate puts this kind of test — match whatever the neighboring code already does), and assert against the specific error variant rather than `is_err()`.\n\n**Shape of the table**\n\nDeclare a small local case struct, then iterate a `Vec` of cases inside one `#[test]`:\n\n```rust\n#[cfg(test)]\nmod tests {\n use super::*;\n\n struct Case {\n name: &'static str,\n input: Order,\n expected: Result<(), OrderError>,\n }\n\n #[test]\n fn validates_order_cases() {\n let cases = vec![\n Case { name: \"happy path\", input: valid_order(), expected: Ok(()) },\n Case {\n name: \"empty line items\",\n input: Order { items: vec![], ..valid_order() },\n expected: Err(OrderError::Empty),\n },\n Case {\n name: \"zero quantity\",\n input: Order { items: vec![item_with_qty(0)], ..valid_order() },\n expected: Err(OrderError::InvalidQuantity { qty: 0 }),\n },\n ];\n\n for case in cases {\n let got = validate_order(&case.input);\n assert_eq!(got, case.expected, \"case: {}\", case.name);\n }\n }\n}\n```\n\n`assert_eq!` works cleanly here if `OrderError` derives `PartialEq` and `Debug` (add `#[derive(Debug, PartialEq)]` if it's your type). The `name` field gives you a labeled failure message so a breaking case is identifiable without a separate test function per case.\n\n**When to split into separate `#[test]` functions instead**\n\nKeep the table for cases that share the same assertion shape. Pull out a separate test when a case needs a different kind of check — e.g. a case that must verify a panic with `#[should_panic]`, or one that asserts on a side effect (a counter, a log) rather than the return value. Don't force unrelated shapes into one table.\n\n**Error-case checks**\n\n- If `OrderError` derives `PartialEq`, `assert_eq!` against the exact variant (as above) is the strongest and simplest choice.\n- If it doesn't (or can't) derive `PartialEq`, use `matches!`:\n\n```rust\nassert!(\n matches!(got, Err(OrderError::InvalidQuantity { qty: 0 })),\n \"case {}: expected InvalidQuantity(0), got {:?}\",\n case.name,\n got\n);\n```\n\n- Never settle for `assert!(got.is_err())`. That passes just as readily for the wrong error variant as for the right one, so it silently hides a regression in the next change.\n\n**Boundary coverage worth including for an order validator**\n\nHappy path (a fully valid order), empty line items, zero quantity, negative quantity, a quantity at the exact allowed maximum, an order with duplicate line items if that's disallowed, and any total/subtotal overflow or minimum-order-value boundary the function documents. Each of these gets its own `Case` row with the precise expected variant, so the whole behavior contract is visible in one place.\n\n**Builders**\n\nFor inputs like `valid_order()` and `item_with_qty(n)`, use small local test helpers (or `..valid_order()` struct update syntax) so the table rows stay readable and each case only states the one field it varies.\n\nOnce written, run `cargo test` and confirm it passes; if a case fails, fix the test unless it has genuinely caught a real bug in `validate_order` — in that case report the bug rather than editing the source while writing tests.",
1674
+ "outputSha256": "8eba00392e2e5b905265d856cd3bfded49aa78bdfd183514b53727884abf3b84",
1675
+ "promptSha256": "c107ab33550812464aa3b85a7a72ac299878517f6f1070b6d77896bd5a11bcb1",
1676
+ "deterministic": [],
1677
+ "judge": {
1678
+ "verdict": "pass",
1679
+ "reason": "Answer shows a Vec<Case> table with input/expected fields iterated in one #[test] (criterion 1), concrete variant assertions via assert_eq!(got, case.expected) with Err(OrderError::Empty)/InvalidQuantity and a matches! fallback (criterion 2), and explains why: 'passes just as readily for the wrong error variant... silently hides a regression' (criterion 3). No fail criterion holds."
1680
+ },
1681
+ "passed": true
1682
+ }
1683
+ ]
1684
+ },
1685
+ {
1686
+ "id": "join-spawned-task",
1687
+ "kind": "behavior",
1688
+ "prompt": "My #[tokio::test] spawns a task with tokio::spawn and I want to wait for it to finish before asserting on the result. What's the right way?",
1689
+ "strictness": "high",
1690
+ "trials": 10,
1691
+ "passes": 10,
1692
+ "passRate": 1,
1693
+ "passAtK": 1,
1694
+ "grader": "judge",
1695
+ "status": "ran",
1696
+ "trialRecords": [
1697
+ {
1698
+ "output": "Join the `JoinHandle` returned by `tokio::spawn` (`.await` it) instead of sleeping. `tokio::spawn` returns a `JoinHandle<T>` that resolves once the task completes:\n\n```rust\n#[tokio::test]\nasync fn spawns_and_completes() {\n let handle = tokio::spawn(async {\n // do work\n 42u32\n });\n\n let result = handle.await.expect(\"task panicked\");\n assert_eq!(result, 42);\n}\n```\n\nNotes:\n\n- The outer `.await` waits for the task to finish. The `Result` from the\n `JoinHandle` is `Err` only if the task *panicked* (or was cancelled) — so\n `.expect(...)`/`?` on it is what surfaces a panic in the spawned task;\n otherwise the test would silently pass. Assert on `Ok(...)`'s inner value\n (or `.unwrap()` it) explicitly.\n- The spawned future must be `Send + 'static` and its output `Send + 'static`.\n If you only need concurrency inside one test without spawning, an inline\n `join!(a, b)` (or `tokio::join!`) avoids the `'static` bound entirely.\n- To check a value produced inside the task without returning it, have the\n task send it over a `tokio::sync::oneshot` (or `mpsc`) channel and `.await`\n the receiver — the channel receive order gives you the same determinism a\n join does.\n- Never `tokio::time::sleep(...).await` \"to let the task finish\": under load\n that either flakes or wastes time, and it can't prove the task actually\n completed. The `JoinHandle` await is the deterministic barrier.\n\nIf the task is spawned inside the code under test rather than the test body,\nhave that code return the `JoinHandle` (or a completion channel/`JoinSet`) so\nthe test can await it — that's a small testability change worth making rather\nthan reaching for a sleep.",
1699
+ "outputSha256": "5eb41ddcf7be3665e04209658bf0c43cfe9da7aff7a265a234fb87bf79d863cb",
1700
+ "promptSha256": "1c342d9565ca22eafaabd90173d1bd5f9776572561e97035ee15835d96bca396",
1701
+ "deterministic": [],
1702
+ "judge": {
1703
+ "verdict": "pass",
1704
+ "reason": "Answer shows concrete join: `let result = handle.await.expect(\"task panicked\");` on the JoinHandle from tokio::spawn before assert_eq. It also names the oneshot/mpsc channel alternative. It explains determinism: 'The JoinHandle await is the deterministic barrier' and warns sleep 'either flakes or wastes time, and it can't prove the task actually completed.' No fixed sleep used as wait."
1705
+ },
1706
+ "passed": true
1707
+ },
1708
+ {
1709
+ "output": "Store the `JoinHandle` returned by `tokio::spawn` and `.await` it before asserting — do not sleep.\n\n```rust\n#[tokio::test]\nasync fn spawned_work_completes() {\n let handle = tokio::spawn(async {\n // ... work that produces the value under test ...\n 42\n });\n\n // Join the task; this awaits actual completion.\n let result = handle.await.expect(\"task panicked\");\n assert_eq!(result, 42);\n}\n```\n\nKey points:\n\n- `tokio::spawn` returns a `JoinHandle<T>`. `handle.await` yields a\n `Result<T, JoinError>`: `Ok(value)` if the task finished normally,\n `Err(JoinError)` if it panicked or was cancelled. Calling `.expect(...)`\n or `matches!` on that `Err` is what turns a silent task panic into a\n visible test failure — a bare `.await` that you ignore would let the\n test pass even if the task blew up.\n- If the task writes to shared state instead of returning a value, use a\n channel: `let (tx, rx) = tokio::sync::oneshot::channel();` (or\n `mpsc`), move `tx` into the task, and `rx.await` in the test before\n asserting. Joining the handle and awaiting the message are both valid;\n pick whichever matches what the task under test actually exposes.\n- If you spawn several tasks, `futures::future::join_all(handles).await`\n (or a `JoinSet`) and assert on the collected results.\n\nNever do this:\n\n```rust\n// WRONG: flaky under load; may assert before the task finishes.\ntokio::spawn(async { /* ... */ });\ntokio::time::sleep(Duration::from_millis(50)).await;\nassert_eq!(shared_value.load(Ordering::SeqCst), 42);\n```\n\nThe sleep is a race, not a synchronization point. A slow CI machine can\nstill lose the race, and a fast one can mask an ordering bug — either way\nthe test result is non-deterministic. Joining the `JoinHandle` (or\nawaiting a channel) establishes a real happens-before relationship, so the\nassertion runs only after the task has actually completed.",
1710
+ "outputSha256": "aed3c2ab1616d76d335cbc1eee5d0d8b4cd53290839bd04ba690c74b6909155a",
1711
+ "promptSha256": "1c342d9565ca22eafaabd90173d1bd5f9776572561e97035ee15835d96bca396",
1712
+ "deterministic": [],
1713
+ "judge": {
1714
+ "verdict": "pass",
1715
+ "reason": "Answer stores the JoinHandle and does `let result = handle.await.expect(\"task panicked\");` before `assert_eq!`, concretely showing the join (criterion 1). It also offers a oneshot channel alternative with `rx.await`. It explains reliability: 'The sleep is a race, not a synchronization point... non-deterministic... Joining the JoinHandle establishes a real happens-before relationship' (criterion 2). The sleep appears only in a 'Never do this' WRONG example, not recommended, so no fail criterion holds."
1716
+ },
1717
+ "passed": true
1718
+ },
1719
+ {
1720
+ "output": "Await the task's `JoinHandle` (or receive from a channel it sends on) — don't sleep.\n\nThe pattern:\n\n```rust\n#[tokio::test]\nasync fn work_produces_expected_result() {\n let handle = tokio::spawn(async_work());\n\n // Join the task. The outer `?`/unwrap is on the JoinError (panic/cancel);\n // the inner value is whatever your task returned.\n let result = handle.await.expect(\"task panicked\");\n\n assert_eq!(result, expected);\n}\n```\n\nKey points:\n\n- `tokio::spawn` returns a `JoinHandle<T>`. `.await`ing it resolves to `Result<T, JoinError>` — `Ok(value)` if the task finished, `Err` if it panicked or was aborted. That \"task panicked\" check is a real assertion you get for free; `sleep`-based waiting loses it entirely.\n- If the task returns a `Result`, you get a nested `Result`: `handle.await.expect(\"task panicked\")?` (or `.unwrap()`) to unwrap both layers before asserting on the value.\n- If you don't actually need a detached concurrent task, the simplest fix is often not to spawn at all — call the `async fn` directly and `.await` it. Only spawn when you specifically need concurrency or a `'static` task.\n- If the task reports its result via a `tokio::sync::mpsc` channel instead of a return value, `rx.recv().await` gives you the deterministic wait and the value in one step; using `JoinHandle` *and* a channel together is redundant.\n\nWhy not `tokio::time::sleep`:\n\n```rust\n// BAD — flaky, non-deterministic\ntokio::spawn(async_work());\ntokio::time::sleep(Duration::from_millis(50)).await; // \"give it time\"\nassert_eq!(shared_state.load(...), expected);\n```\n\nThis passes or fails depending on machine load and scheduler timing; under a busy CI runner 50ms may not be enough. Joining the handle is exact: the test proceeds the instant the task is done, and blocks until then.\n\nOne caveat: `#[tokio::test]` uses a single-threaded runtime by default, which is fine for `.await`ing a `JoinHandle` (the task gets polled cooperatively). If your task does blocking work that never yields, `JoinHandle.await` will deadlock on the current-thread runtime — use `#[tokio::test(flavor = \"multi_thread\", worker_threads = 2)]`, or better, move genuinely blocking work to `spawn_blocking`.",
1721
+ "outputSha256": "149c02ba8f0b5d34872625d3d2d02a84d9cdf687853e3cd383a79db037416eee",
1722
+ "promptSha256": "1c342d9565ca22eafaabd90173d1bd5f9776572561e97035ee15835d96bca396",
1723
+ "deterministic": [],
1724
+ "judge": {
1725
+ "verdict": "pass",
1726
+ "reason": "Answer shows `let handle = tokio::spawn(async_work()); let result = handle.await.expect(\"task panicked\");` then asserts — concrete JoinHandle .await join before assertion (crit 1). It explains sleep is flaky/non-deterministic, passing/failing on machine load, while joining is exact and proceeds the instant the task is done (crit 2). No fixed sleep used as wait; the sleep example is explicitly labeled BAD."
1727
+ },
1728
+ "passed": true
1729
+ },
1730
+ {
1731
+ "output": "Await the `JoinHandle` returned by `tokio::spawn`. That's the whole answer, and it's the only thing that's deterministic:\n\n```rust\n#[tokio::test]\nasync fn background_task_updates_state() {\n let handle = tokio::spawn(async move {\n do_work().await;\n 41\n });\n\n // Join the handle. This resolves only after the task has completed.\n let result = handle.await.expect(\"task panicked\");\n\n assert_eq!(result, 41);\n}\n```\n\nKey points:\n\n- `handle.await` is a `Result<T, JoinError>`. You get `T` (the task body's return value) on success. `JoinError` means the task panicked or was cancelled — `.expect(...)` on it surfaces the panic instead of silently passing.\n- If the task's *only* job is to produce a value, don't spawn at all — just `.await` the future directly. Spawning is only needed when you genuinely want concurrency.\n- To run multiple tasks concurrently, spawn them first, collect the handles into a `Vec`, then join them all:\n\n```rust\nlet handles: Vec<_> = (0..4).map(|i| tokio::spawn(work(i))).collect();\n\nfor handle in handles {\n handle.await.expect(\"task panicked\");\n}\n```\n\n- If the task communicates via a channel rather than a return value, wait on the channel's `recv().await` (`None` indicates the sender was dropped, i.e. the task finished) instead of sleeping.\n\nNever do this:\n\n```rust\ntokio::spawn(async { /* ... */ });\ntokio::time::sleep(Duration::from_millis(50)).await; // flaky under load\nassert_eq!(shared.load(Ordering::SeqCst), 41);\n```\n\nThe sleep introduces a race: it can pass on a fast dev machine and fail under CI load, and it will silently pass even if the task never ran at all.\n\nOne more thing worth knowing: with `#[tokio::test]` the default runtime is single-threaded, so a spawned task only makes progress when your test body awaits. Joining the handle provides that await point, so this pattern works correctly on the default test runtime too.",
1732
+ "outputSha256": "c4e2a9c9fbc88a25829ca3ae5f8ba6df89f7c86b78a627e931e66c3ad1150527",
1733
+ "promptSha256": "1c342d9565ca22eafaabd90173d1bd5f9776572561e97035ee15835d96bca396",
1734
+ "deterministic": [],
1735
+ "judge": {
1736
+ "verdict": "pass",
1737
+ "reason": "Answer shows `let result = handle.await.expect(\"task panicked\");` on the tokio::spawn JoinHandle before assert_eq, and channel recv().await alternative. Explains determinism: sleep 'introduces a race... can pass on a fast dev machine and fail under CI load.' Sleep shown only as anti-pattern to avoid, not recommended."
1738
+ },
1739
+ "passed": true
1740
+ },
1741
+ {
1742
+ "output": "Await the task's `JoinHandle`, don't sleep.\n\n```rust\n#[tokio::test]\nasync fn spawned_task_updates_state() {\n let handle = tokio::spawn(async move {\n // ... do the work, return the value to assert on\n compute()\n });\n\n // .await on the JoinHandle waits for the task to finish.\n // It returns Result<T, JoinError>, so unwrap the join too.\n let result = handle.await.expect(\"task panicked\");\n assert_eq!(result, expected);\n}\n```\n\nKey points:\n\n- **`.await` the `JoinHandle`**, which resolves once the spawned task completes (or panics — hence the `JoinError` you unwrap). This is deterministic; a `tokio::time::sleep(Duration::from_millis(50))` is not and will flake under load.\n- **Return the value you want to assert on from the task closure**, so the `JoinHandle`'s output carries it back. If the task mutates shared state instead (e.g. an `Arc<Mutex<..>>`), that's fine too — just `handle.await.expect(...)` to synchronize before locking and asserting, rather than sleeping.\n- **If there are several tasks**, collect their handles and await them all before asserting:\n\n```rust\nlet mut handles = Vec::new();\nfor input in inputs {\n handles.push(tokio::spawn(async move { process(input) }));\n}\n\nlet mut results = Vec::new();\nfor handle in handles {\n results.push(handle.await.expect(\"task panicked\"));\n}\nassert_eq!(results, expected);\n```\n\n- **If the task can't easily return a value** (long-lived background loop, etc.), have it send its final result over a `tokio::sync::oneshot`/`mpsc` channel and `recv().await` on the receiving end before asserting — again, no sleep.\n\nOne thing to watch: `#[tokio::test]` defaults to a current-thread runtime, which is fine for awaiting a `JoinHandle`. If your spawned task depends on real concurrency across threads (e.g. blocking work, or `tokio::task::spawn_blocking` expectations), use `#[tokio::test(flavor = \"multi_thread\")]` so it actually runs on the multi-threaded scheduler.",
1743
+ "outputSha256": "2f0cc07f84d41c4740bac590ff01e83a08a92916b20d23e509a74f1e9854648f",
1744
+ "promptSha256": "1c342d9565ca22eafaabd90173d1bd5f9776572561e97035ee15835d96bca396",
1745
+ "deterministic": [],
1746
+ "judge": {
1747
+ "verdict": "pass",
1748
+ "reason": "Answer shows `let result = handle.await.expect(\"task panicked\");` on the JoinHandle before `assert_eq!`, and a multi-task loop awaiting each handle. It explains determinism: '.await the JoinHandle ... This is deterministic; a tokio::time::sleep(...) is not and will flake under load.' No fixed sleep used as a wait mechanism."
1749
+ },
1750
+ "passed": true
1751
+ },
1752
+ {
1753
+ "output": "Await the task's `JoinHandle` (or a channel) — never `tokio::time::sleep` to \"give it time.\"\n\n```rust\n#[tokio::test]\nasync fn spawned_task_produces_expected_result() {\n let handle = tokio::spawn(async {\n // ...work...\n 42\n });\n\n // JoinHandle itself is a future; .await yields Result<T, JoinError>\n let result = handle.await.expect(\"task panicked or was cancelled\");\n assert_eq!(result, 42);\n}\n```\n\nKey points:\n\n- **`handle.await` returns `Result<T, JoinError>`.** The `Ok` variant is your task's value; the `Err` variant means the task panicked or was cancelled. Use `.expect(...)` (or match and assert the variant explicitly) so a panicking task fails the test with a clear message instead of being masked.\n\n- **If the task moves data out** and you'd rather not thread it back through `JoinHandle`, use a channel and await the receive:\n\n ```rust\n let (tx, rx) = tokio::sync::oneshot::channel();\n tokio::spawn(async move {\n let value = do_work().await;\n let _ = tx.send(value);\n });\n\n let value = rx.await.expect(\"task dropped the sender without sending\");\n assert_eq!(value, expected);\n ```\n\n `rx.await` resolves once the sender sends or is dropped, so the assertion runs after the work completes — deterministically, no timing involved.\n\n- **Don't do this:**\n\n ```rust\n let handle = tokio::spawn(async { /* ... */ });\n tokio::time::sleep(Duration::from_millis(50)).await; // ✗ flaky under load\n // now \"assume\" the task finished\n ```\n\n A fixed sleep is a race: the task may not have finished, and under a loaded CI machine it can take longer than the sleep, so the test flakes or passes for the wrong reason.\n\n- **If you only need the assertion, skip the spawn.** `#[tokio::test]` gives you a runtime; you can `do_work().await` directly. Spawning is worth it when you're specifically testing concurrency (e.g., two tasks racing, or a task that outlives the current scope). For a plain \"run this async function and check the result,\" there's nothing to join:\n\n ```rust\n #[tokio::test]\n async fn direct_await_is_simpler() {\n let result = do_work().await;\n assert_eq!(result, expected);\n }\n ```\n\n- **Never `handle.abort()` and then assert on the value** unless you actually intend to test cancellation — you'll get a `JoinError::Cancelled` and the value is gone.\n\n- **For untrusted/structured input, consider `proptest`** to generate cases rather than hand-writing many near-identical tests, but keep the join discipline the same inside generated cases.\n\nSo: `handle.await` for a `JoinHandle`, `rx.await` for a channel — both make the test wait on the actual completion signal rather than a guessed wall-clock delay.",
1754
+ "outputSha256": "dc0ab185dd8db984ea86616903762d6591f0ebf227f7ccca833c4b7304ce69d1",
1755
+ "promptSha256": "1c342d9565ca22eafaabd90173d1bd5f9776572561e97035ee15835d96bca396",
1756
+ "deterministic": [],
1757
+ "judge": {
1758
+ "verdict": "pass",
1759
+ "reason": "Answer shows concrete `let result = handle.await.expect(...)` on the tokio::spawn JoinHandle before assert_eq, plus a oneshot channel `rx.await` alternative. It explains determinism: 'no timing involved', fixed sleep is 'a race... under a loaded CI machine it can take longer... flakes'. Sleep is only shown as a warned-against anti-pattern, not recommended."
1760
+ },
1761
+ "passed": true
1762
+ },
1763
+ {
1764
+ "output": "`.await` the task's `JoinHandle` directly, then assert on the result.\n\n```rust\n#[tokio::test]\nasync fn processes_order() {\n let handle = tokio::spawn(async move {\n process_order(42).await\n });\n\n // Join the task and unwrap the two layers:\n // JoinError from the task itself, then the fn's Result.\n let result = handle.await.expect(\"task panicked\");\n assert_eq!(result, Ok(OrderStatus::Complete));\n}\n```\n\nThe shape of `.await` on a `JoinHandle<T>` is `Result<T, JoinError>`:\n- `JoinError` means the task panicked or was cancelled — `.expect(...)` (or an explicit match) surfaces that as a test failure.\n- The inner `T` is whatever the spawned async block returned — if that's itself a `Result`, match/`assert_eq!` on the specific variant, not just `is_ok()`.\n\n**If you don't need the task to run concurrently with the test body**, don't spawn at all — just `.await` the future inline; spawning only makes sense when you're actually testing concurrency:\n\n```rust\n#[tokio::test]\nasync fn processes_order() {\n let result = process_order(42).await;\n assert_eq!(result, Ok(OrderStatus::Complete));\n}\n```\n\n**If you're testing multiple tasks**, join them all before asserting:\n\n```rust\nlet handles: Vec<_> = (0..4).map(|i| tokio::spawn(process_order(i))).collect();\nlet results: Vec<_> = futures::future::join_all(handles).await\n .into_iter()\n .map(|r| r.expect(\"task panicked\"))\n .collect();\n```\n\n**What to avoid:** never `tokio::time::sleep(Duration::from_millis(50)).await` to \"give it time to finish.\" That's non-deterministic — it flakes under load and passes before the task is actually done on a fast machine, and it's explicitly flagged in this skill's rules. The `JoinHandle` is the synchronization primitive; a sleep is a guess at one.\n\nA couple of related notes:\n- If the task talks back over a channel, the receiver side can also be your synchronization point — drain the channel until it closes (`while let Some(msg) = rx.recv().await { ... }`), which naturally ends when the sender task drops.\n- With `#[tokio::test]`, use the current-thread runtime (the default) unless the test genuinely needs multi-thread scheduling. `tokio::spawn` works on the current-thread runtime too, so this rarely matters — but if you're specifically testing multi-threaded behavior, that's the one case to reach for `#[tokio::test(flavor = \"multi_thread\")]`.",
1765
+ "outputSha256": "f3b797ea5d578ba7e2c5360dcd86f757658c667a2327a54f78db0e557fa0d7ac",
1766
+ "promptSha256": "1c342d9565ca22eafaabd90173d1bd5f9776572561e97035ee15835d96bca396",
1767
+ "deterministic": [],
1768
+ "judge": {
1769
+ "verdict": "pass",
1770
+ "reason": "Answer shows `let result = handle.await.expect(\"task panicked\");` on the JoinHandle from tokio::spawn before asserting, and join_all for multiple tasks. It explains reliability: sleep is 'non-deterministic — it flakes under load and passes before the task is actually done'; 'The JoinHandle is the synchronization primitive; a sleep is a guess at one.' No fixed sleep used as wait."
1771
+ },
1772
+ "passed": true
1773
+ },
1774
+ {
1775
+ "output": "Await the task's `JoinHandle` — don't sleep.\n\n```rust\n#[tokio::test]\nasync fn worker_updates_state() {\n let handle = tokio::spawn(async move {\n do_work().await;\n 42 // task return value\n });\n\n // Await the JoinHandle itself; this resolves when the task completes.\n let result = handle.await.expect(\"task panicked\");\n assert_eq!(result, 42);\n}\n```\n\nKey points:\n\n- **`tokio::spawn` returns a `JoinHandle`, and `.await` on it yields `Result<T, JoinError>`** where `T` is the task's return type. The future resolves exactly when the spawned task finishes, so there's no race.\n- **The `Result` is because a spawned task can panic.** Unwrapping (or better, `.expect(\"...\")` with a message) turns a panic inside the task into a test failure at the await point instead of a silently-passing or hanging test. If a panicking task should be an assertion rather than a bug, match on the `JoinError` explicitly.\n- **If you only need the side effect and not a value**, have the task return `()` and do `handle.await.expect(\"task panicked\");` before asserting on the shared state.\n- **If the task communicates via a channel**, you can await `rx.recv()` instead — but you still must ensure the task is driven to completion (e.g. by the channel closing when it finishes), not by a timer.\n- **Multiple tasks:** `futures::future::join_all(handles).await` (or `tokio::try_join!` for a fixed few) awaits all of them and collects results.\n\nWhat *not* to do:\n\n```rust\n// Bad: flaky, non-deterministic under load.\ntokio::spawn(async { do_work().await; });\ntokio::time::sleep(std::time::Duration::from_millis(50)).await;\nassert_eq!(get_state(), 42);\n```\n\nThe sleep guesses how long the task needs; under a loaded CI runner that guess is wrong and the test flakes. Awaiting the `JoinHandle` is deterministic — it waits precisely until the task is done, no more, no less.\n\nOne more note: `#[tokio::test]` defaults to a single-threaded runtime. Tasks spawned with `tokio::spawn` are still driven on that runtime as your test future yields at the `.await` on the handle, so the pattern above works fine — but if the spawned task blocks (does synchronous CPU/blocking I/O) rather than yielding at `await` points, use `#[tokio::test(flavor = \"multi_thread\")]` or `spawn_blocking` so it can actually make progress.",
1776
+ "outputSha256": "5080922187688dde7603bc8e7581f2913cd5838edf473479bd2669d53c7021c4",
1777
+ "promptSha256": "1c342d9565ca22eafaabd90173d1bd5f9776572561e97035ee15835d96bca396",
1778
+ "deterministic": [],
1779
+ "judge": {
1780
+ "verdict": "pass",
1781
+ "reason": "Answer shows `let handle = tokio::spawn(...)` then `let result = handle.await.expect(\"task panicked\");` before `assert_eq!`, satisfying criterion 1 concretely. It explains determinism: 'The future resolves exactly when the spawned task finishes, so there's no race' and 'Awaiting the JoinHandle is deterministic — it waits precisely until the task is done.' The sleep example is shown only as a 'Bad' anti-pattern to avoid, not recommended, so no fail criterion holds."
1782
+ },
1783
+ "passed": true
1784
+ },
1785
+ {
1786
+ "output": "Await the task's `JoinHandle` — that's what `tokio::spawn` gives back, and awaiting it both drives the test to wait for the task and surfaces a panic from inside the task as a `JoinError` instead of a hang.\n\n```rust\n#[tokio::test]\nasync fn spawned_task_completes_with_expected_result() {\n let handle = tokio::spawn(async {\n // ... work ...\n 42\n });\n\n let value = handle.await.expect(\"task panicked\");\n assert_eq!(value, 42);\n}\n```\n\nKey points:\n\n- `handle.await` returns `Result<T, JoinError>`. The `Ok` case is the task's return value; the `Err` case means the task panicked or was cancelled. Use `.expect(...)` (or assert on the error) so a panicking task fails the test loudly rather than passing silently.\n- If the spawned task communicates via a channel instead of a return value, await the receiver (e.g. `rx.recv().await`) — that is still a proper join, whereas sleeping is not.\n- If several tasks are spawned, collect their handles and await them all (join them one by one, or with `futures::future::join_all` if you have `futures` as a dev-dependency) before asserting.\n\nDo not do this:\n\n```rust\ntokio::spawn(async { /* ... */ });\ntokio::time::sleep(Duration::from_millis(50)).await; // flaky under load\nassert_eq!(shared_flag.load(Ordering::SeqCst), true);\n```\n\nA sleep \"waits\" only by luck; under a loaded CI machine 50ms may not be enough, and you get a nondeterministic failure. Awaiting the `JoinHandle` has no such race — the test resumes exactly when the task completes (or panics).",
1787
+ "outputSha256": "5bd6d11a31f39b810f826e1b0abf66221b40a79b84c2fcf76a2660e298fdee8a",
1788
+ "promptSha256": "1c342d9565ca22eafaabd90173d1bd5f9776572561e97035ee15835d96bca396",
1789
+ "deterministic": [],
1790
+ "judge": {
1791
+ "verdict": "pass",
1792
+ "reason": "Answer shows concrete join: `let handle = tokio::spawn(...)` then `let value = handle.await.expect(\"task panicked\")` before `assert_eq!`. Also names channel alternative (`rx.recv().await`). Explains determinism: 'the test resumes exactly when the task completes', sleep 'waits only by luck' and is flaky under load. Sleep shown only as anti-pattern to avoid."
1793
+ },
1794
+ "passed": true
1795
+ },
1796
+ {
1797
+ "output": "Await the task's `JoinHandle` before asserting. `tokio::spawn` returns a `JoinHandle<T>`, and awaiting it gives back `Result<T, JoinError>` — `Ok(value)` if the task ran to completion, `Err` if it panicked or was cancelled.\n\n```rust\n#[tokio::test]\nasync fn spawned_task_updates_state() {\n let handle = tokio::spawn(async move {\n // ... do the work ...\n 42\n });\n\n // Join the handle so the task is guaranteed finished before asserting.\n let result = handle.await.expect(\"task panicked\");\n assert_eq!(result, 42);\n}\n```\n\nA few things worth getting right:\n\n- **Don't** use `tokio::time::sleep(Duration::from_millis(50)).await` to \"give the task time to finish.\" It's non-deterministic under CI load and will eventually flake. Joining the handle is deterministic.\n- **Only relevant with the multi-threaded runtime:** with `#[tokio::test]` (current-thread runtime) tasks still get driven by the runtime while you `.await` the handle, so `.await`ing the handle is sufficient either way — no extra `yield_now` needed.\n- **If the task returns `()`**, `handle.await` is still `Result<(), JoinError>`; unwrap it (or `assert!(handle.await.is_ok())`) so a panic inside the spawned task fails the test instead of being silently dropped.\n- **Multiple tasks:** collect the handles (`let handles: Vec<_> = ...`) and `for h in handles { h.await.unwrap(); }`, or use `futures::future::join_all(handles).await` if you've got the `futures` crate. Note `tokio::join!` is for futures you already have inline, not for handles in a `Vec`.\n- **If the spawned task sends its result over a `tokio::sync::mpsc`/`oneshot` channel** instead of returning it, awaiting the receiver is the equivalent synchronization point — e.g. `rx.await.unwrap()` — and you can skip joining the handle if you don't care about the task itself panicking (though asserting on the channel's `RecvError` covers the \"task dropped its sender without sending\" case).\n\nSo: prefer `handle.await` (or awaiting the channel the task writes to) as the \"the task is done\" signal, and treat any sleep-based wait as a bug in the test.",
1798
+ "outputSha256": "89f3ff6aae08e1c3a06a2da4b35fff139b28e7d7b8e4ffd7ca6dbd2865ba4338",
1799
+ "promptSha256": "1c342d9565ca22eafaabd90173d1bd5f9776572561e97035ee15835d96bca396",
1800
+ "deterministic": [],
1801
+ "judge": {
1802
+ "verdict": "pass",
1803
+ "reason": "Answer shows concrete `let result = handle.await.expect(\"task panicked\");` on the JoinHandle before `assert_eq!`, and channel alternative `rx.await.unwrap()`. It explains determinism: sleep is 'non-deterministic under CI load and will eventually flake. Joining the handle is deterministic.' No fixed sleep used as wait; sleep only warned against."
1804
+ },
1805
+ "passed": true
1806
+ }
1807
+ ]
1808
+ }
1809
+ ],
1810
+ "verdict": "fail",
1811
+ "scope": "bundled",
1812
+ "skillDigest": "2e16c0da7abdd208e3f8ef511f07266ef7072c9cc94bbaf72c1a6f624263b05b",
1813
+ "catalogDigest": "d09b13e321c66a435263da60f337e323760ef9d3d394b30d1ab3f41817a01f39",
1814
+ "judgePromptVersion": "2026-09-25.1",
1815
+ "runner": "deepseek",
1816
+ "model": "deepseek-chat",
1817
+ "runnerPromptVersion": "2026-09-25.1",
1818
+ "recordedAt": "2026-09-25T20:55:09.629Z",
1819
+ "judge": "deepseek",
1820
+ "judgeModel": "deepseek-chat"
1821
+ }
1822
+ ]
1823
+ }