@mrciphersmith/keryx 0.2.164 → 0.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (182) hide show
  1. package/README.md +4 -1
  2. package/dist/cli.js +82540 -50300
  3. package/dist/core.js +28967 -18937
  4. package/package.json +2 -2
  5. package/src/gdgraph/affected-report.ts +141 -0
  6. package/src/gdgraph/build.ts +170 -23
  7. package/src/gdgraph/service.ts +6 -0
  8. package/src/gdgraph/staleness.ts +253 -45
  9. package/src/gdskills/bundled/agents/codebase-navigator.md +55 -0
  10. package/src/gdskills/bundled/agents/design-advisor.md +64 -0
  11. package/src/gdskills/bundled/agents/docs-maintainer.md +56 -0
  12. package/src/gdskills/bundled/agents/end-to-end-tester.md +56 -0
  13. package/src/gdskills/bundled/agents/error-path-auditor.md +57 -0
  14. package/src/gdskills/bundled/agents/go-build-fixer.md +52 -0
  15. package/src/gdskills/bundled/agents/go-code-auditor.md +49 -0
  16. package/src/gdskills/bundled/agents/performance-auditor.md +63 -0
  17. package/src/gdskills/bundled/agents/python-build-fixer.md +52 -0
  18. package/src/gdskills/bundled/agents/python-code-auditor.md +49 -0
  19. package/src/gdskills/bundled/agents/refactoring-steward.md +61 -0
  20. package/src/gdskills/bundled/agents/security-auditor.md +62 -0
  21. package/src/gdskills/bundled/agents/test-first-driver.md +61 -0
  22. package/src/gdskills/bundled/agents/work-planner.md +62 -0
  23. package/src/gdskills/bundled/install-manifest.json +797 -0
  24. package/src/gdskills/bundled/rules/core/model-selection.mdc +51 -0
  25. package/src/gdskills/bundled/rules/core/skill-lifecycle.mdc +29 -1
  26. package/src/gdskills/bundled/rules/core/skills-storage-workflow.mdc +2 -2
  27. package/src/gdskills/bundled/skills/review/code-style-review/SKILL.md +1 -1
  28. package/src/gdskills/bundled/skills/review/review-jev-rules/SKILL.md +267 -0
  29. package/src/gdskills/bundled/skills/review/review-orchestrator/SKILL.detail.md +26 -0
  30. package/src/gdskills/bundled/skills/review/review-orchestrator/SKILL.md +75 -247
  31. package/src/gdskills/bundled/skills/review/review-orchestrator/output-contract.schema.json +19 -0
  32. package/src/gdskills/bundled/skills/review/review-orchestrator/reviewer-finding.schema.json +10 -0
  33. package/src/gdskills/bundled/skills/review/review-orchestrator/reviewer-input.schema.json +5 -0
  34. package/src/gdskills/bundled/skills/review/review-orchestrator/templates/pr-comment-backend.md +50 -0
  35. package/src/gdskills/bundled/skills/review/review-orchestrator/templates/pr-comment-frontend.md +52 -0
  36. package/src/gdskills/bundled/skills/review/review-orchestrator/templates/review-report.md +143 -0
  37. package/src/gdskills/bundled/stacks/angular/agent-refs.json +4 -0
  38. package/src/gdskills/bundled/stacks/angular/governance/eval.json +1751 -0
  39. package/src/gdskills/bundled/stacks/angular/governance/scout.json +32 -0
  40. package/src/gdskills/bundled/stacks/angular/pack.json +55 -0
  41. package/src/gdskills/bundled/stacks/angular/rules/coding-style.mdc +82 -0
  42. package/src/gdskills/bundled/stacks/angular/rules/patterns.mdc +84 -0
  43. package/src/gdskills/bundled/stacks/angular/rules/security.mdc +70 -0
  44. package/src/gdskills/bundled/stacks/angular/rules/testing.mdc +73 -0
  45. package/src/gdskills/bundled/stacks/angular/skills/angular-build-fix/SKILL.md +127 -0
  46. package/src/gdskills/bundled/stacks/angular/skills/angular-build-fix/evals.json +72 -0
  47. package/src/gdskills/bundled/stacks/angular/skills/angular-code-review/SKILL.md +98 -0
  48. package/src/gdskills/bundled/stacks/angular/skills/angular-code-review/evals.json +73 -0
  49. package/src/gdskills/bundled/stacks/angular/skills/angular-implementation/SKILL.md +112 -0
  50. package/src/gdskills/bundled/stacks/angular/skills/angular-implementation/evals.json +74 -0
  51. package/src/gdskills/bundled/stacks/angular/skills/angular-testing/SKILL.md +102 -0
  52. package/src/gdskills/bundled/stacks/angular/skills/angular-testing/evals.json +71 -0
  53. package/src/gdskills/bundled/stacks/go/agent-refs.json +3 -0
  54. package/src/gdskills/bundled/stacks/go/governance/eval.json +1745 -0
  55. package/src/gdskills/bundled/stacks/go/governance/scout.json +31 -0
  56. package/src/gdskills/bundled/stacks/go/pack.json +41 -0
  57. package/src/gdskills/bundled/stacks/go/rules/coding-style.mdc +85 -0
  58. package/src/gdskills/bundled/stacks/go/rules/patterns.mdc +65 -0
  59. package/src/gdskills/bundled/stacks/go/rules/security.mdc +73 -0
  60. package/src/gdskills/bundled/stacks/go/rules/testing.mdc +68 -0
  61. package/src/gdskills/bundled/stacks/go/skills/go-build-fix/SKILL.md +138 -0
  62. package/src/gdskills/bundled/stacks/go/skills/go-build-fix/evals.json +75 -0
  63. package/src/gdskills/bundled/stacks/go/skills/go-code-review/SKILL.md +121 -0
  64. package/src/gdskills/bundled/stacks/go/skills/go-code-review/evals.json +72 -0
  65. package/src/gdskills/bundled/stacks/go/skills/go-implementation/SKILL.md +122 -0
  66. package/src/gdskills/bundled/stacks/go/skills/go-implementation/evals.json +76 -0
  67. package/src/gdskills/bundled/stacks/go/skills/go-testing/SKILL.md +126 -0
  68. package/src/gdskills/bundled/stacks/go/skills/go-testing/evals.json +73 -0
  69. package/src/gdskills/bundled/stacks/mobx/agent-refs.json +4 -0
  70. package/src/gdskills/bundled/stacks/mobx/governance/eval.json +904 -0
  71. package/src/gdskills/bundled/stacks/mobx/governance/scout.json +18 -0
  72. package/src/gdskills/bundled/stacks/mobx/pack.json +28 -0
  73. package/src/gdskills/bundled/stacks/mobx/rules/coding-style.mdc +91 -0
  74. package/src/gdskills/bundled/stacks/mobx/rules/patterns.mdc +122 -0
  75. package/src/gdskills/bundled/stacks/mobx/rules/security.mdc +56 -0
  76. package/src/gdskills/bundled/stacks/mobx/rules/testing.mdc +63 -0
  77. package/src/gdskills/bundled/stacks/mobx/skills/mobx-observable-testing/SKILL.md +124 -0
  78. package/src/gdskills/bundled/stacks/mobx/skills/mobx-observable-testing/evals.json +73 -0
  79. package/src/gdskills/bundled/stacks/mobx/skills/mobx-store-implementation/SKILL.md +149 -0
  80. package/src/gdskills/bundled/stacks/mobx/skills/mobx-store-implementation/evals.json +74 -0
  81. package/src/gdskills/bundled/stacks/nestjs/agent-refs.json +4 -0
  82. package/src/gdskills/bundled/stacks/nestjs/governance/eval.json +1308 -0
  83. package/src/gdskills/bundled/stacks/nestjs/governance/scout.json +34 -0
  84. package/src/gdskills/bundled/stacks/nestjs/pack.json +53 -0
  85. package/src/gdskills/bundled/stacks/nestjs/rules/coding-style.mdc +70 -0
  86. package/src/gdskills/bundled/stacks/nestjs/rules/patterns.mdc +83 -0
  87. package/src/gdskills/bundled/stacks/nestjs/rules/security.mdc +73 -0
  88. package/src/gdskills/bundled/stacks/nestjs/rules/testing.mdc +69 -0
  89. package/src/gdskills/bundled/stacks/nestjs/skills/nestjs-build-fix/SKILL.md +157 -0
  90. package/src/gdskills/bundled/stacks/nestjs/skills/nestjs-build-fix/evals.json +70 -0
  91. package/src/gdskills/bundled/stacks/nestjs/skills/nestjs-implementation/SKILL.md +129 -0
  92. package/src/gdskills/bundled/stacks/nestjs/skills/nestjs-implementation/evals.json +71 -0
  93. package/src/gdskills/bundled/stacks/nestjs/skills/nestjs-testing/SKILL.md +143 -0
  94. package/src/gdskills/bundled/stacks/nestjs/skills/nestjs-testing/evals.json +69 -0
  95. package/src/gdskills/bundled/stacks/nextjs-nuxt/agent-refs.json +4 -0
  96. package/src/gdskills/bundled/stacks/nextjs-nuxt/governance/eval.json +2413 -0
  97. package/src/gdskills/bundled/stacks/nextjs-nuxt/governance/scout.json +42 -0
  98. package/src/gdskills/bundled/stacks/nextjs-nuxt/pack.json +42 -0
  99. package/src/gdskills/bundled/stacks/nextjs-nuxt/rules/coding-style.mdc +69 -0
  100. package/src/gdskills/bundled/stacks/nextjs-nuxt/rules/patterns.mdc +88 -0
  101. package/src/gdskills/bundled/stacks/nextjs-nuxt/rules/security.mdc +72 -0
  102. package/src/gdskills/bundled/stacks/nextjs-nuxt/rules/testing.mdc +64 -0
  103. package/src/gdskills/bundled/stacks/nextjs-nuxt/skills/nextjs-nuxt-build-fix/SKILL.md +147 -0
  104. package/src/gdskills/bundled/stacks/nextjs-nuxt/skills/nextjs-nuxt-build-fix/evals.json +75 -0
  105. package/src/gdskills/bundled/stacks/nextjs-nuxt/skills/nextjs-nuxt-code-review/SKILL.md +118 -0
  106. package/src/gdskills/bundled/stacks/nextjs-nuxt/skills/nextjs-nuxt-code-review/evals.json +76 -0
  107. package/src/gdskills/bundled/stacks/nextjs-nuxt/skills/nextjs-nuxt-implementation/SKILL.md +135 -0
  108. package/src/gdskills/bundled/stacks/nextjs-nuxt/skills/nextjs-nuxt-implementation/evals.json +78 -0
  109. package/src/gdskills/bundled/stacks/nextjs-nuxt/skills/nextjs-nuxt-testing/SKILL.md +116 -0
  110. package/src/gdskills/bundled/stacks/nextjs-nuxt/skills/nextjs-nuxt-testing/evals.json +75 -0
  111. package/src/gdskills/bundled/stacks/nextjs-nuxt/skills/nextjs-nuxt-upgrade-migration/SKILL.md +134 -0
  112. package/src/gdskills/bundled/stacks/nextjs-nuxt/skills/nextjs-nuxt-upgrade-migration/evals.json +76 -0
  113. package/src/gdskills/bundled/stacks/python/agent-refs.json +3 -0
  114. package/src/gdskills/bundled/stacks/python/governance/eval.json +1758 -0
  115. package/src/gdskills/bundled/stacks/python/governance/scout.json +34 -0
  116. package/src/gdskills/bundled/stacks/python/pack.json +41 -0
  117. package/src/gdskills/bundled/stacks/python/rules/coding-style.mdc +63 -0
  118. package/src/gdskills/bundled/stacks/python/rules/patterns.mdc +88 -0
  119. package/src/gdskills/bundled/stacks/python/rules/security.mdc +84 -0
  120. package/src/gdskills/bundled/stacks/python/rules/testing.mdc +77 -0
  121. package/src/gdskills/bundled/stacks/python/skills/python-build-fix/SKILL.md +144 -0
  122. package/src/gdskills/bundled/stacks/python/skills/python-build-fix/evals.json +74 -0
  123. package/src/gdskills/bundled/stacks/python/skills/python-code-review/SKILL.md +155 -0
  124. package/src/gdskills/bundled/stacks/python/skills/python-code-review/evals.json +72 -0
  125. package/src/gdskills/bundled/stacks/python/skills/python-implementation/SKILL.md +143 -0
  126. package/src/gdskills/bundled/stacks/python/skills/python-implementation/evals.json +78 -0
  127. package/src/gdskills/bundled/stacks/python/skills/python-testing/SKILL.md +132 -0
  128. package/src/gdskills/bundled/stacks/python/skills/python-testing/evals.json +73 -0
  129. package/src/gdskills/bundled/stacks/react/agent-refs.json +4 -0
  130. package/src/gdskills/bundled/stacks/react/governance/eval.json +2188 -0
  131. package/src/gdskills/bundled/stacks/react/governance/scout.json +40 -0
  132. package/src/gdskills/bundled/stacks/react/pack.json +42 -0
  133. package/src/gdskills/bundled/stacks/react/rules/coding-style.mdc +58 -0
  134. package/src/gdskills/bundled/stacks/react/rules/patterns.mdc +79 -0
  135. package/src/gdskills/bundled/stacks/react/rules/security.mdc +70 -0
  136. package/src/gdskills/bundled/stacks/react/rules/testing.mdc +60 -0
  137. package/src/gdskills/bundled/stacks/react/skills/react-build-fix/SKILL.md +139 -0
  138. package/src/gdskills/bundled/stacks/react/skills/react-build-fix/evals.json +72 -0
  139. package/src/gdskills/bundled/stacks/react/skills/react-code-review/SKILL.md +148 -0
  140. package/src/gdskills/bundled/stacks/react/skills/react-code-review/evals.json +74 -0
  141. package/src/gdskills/bundled/stacks/react/skills/react-implementation/SKILL.md +140 -0
  142. package/src/gdskills/bundled/stacks/react/skills/react-implementation/evals.json +74 -0
  143. package/src/gdskills/bundled/stacks/react/skills/react-testing/SKILL.md +142 -0
  144. package/src/gdskills/bundled/stacks/react/skills/react-testing/evals.json +83 -0
  145. package/src/gdskills/bundled/stacks/react/skills/react-upgrade-migration/SKILL.md +155 -0
  146. package/src/gdskills/bundled/stacks/react/skills/react-upgrade-migration/evals.json +74 -0
  147. package/src/gdskills/bundled/stacks/ts-js-node/agent-refs.json +4 -0
  148. package/src/gdskills/bundled/stacks/ts-js-node/governance/eval.json +2155 -0
  149. package/src/gdskills/bundled/stacks/ts-js-node/governance/scout.json +40 -0
  150. package/src/gdskills/bundled/stacks/ts-js-node/pack.json +41 -0
  151. package/src/gdskills/bundled/stacks/ts-js-node/rules/coding-style.mdc +73 -0
  152. package/src/gdskills/bundled/stacks/ts-js-node/rules/patterns.mdc +61 -0
  153. package/src/gdskills/bundled/stacks/ts-js-node/rules/security.mdc +71 -0
  154. package/src/gdskills/bundled/stacks/ts-js-node/rules/testing.mdc +63 -0
  155. package/src/gdskills/bundled/stacks/ts-js-node/skills/nodejs-build-fix/SKILL.md +137 -0
  156. package/src/gdskills/bundled/stacks/ts-js-node/skills/nodejs-build-fix/evals.json +73 -0
  157. package/src/gdskills/bundled/stacks/ts-js-node/skills/nodejs-code-review/SKILL.md +124 -0
  158. package/src/gdskills/bundled/stacks/ts-js-node/skills/nodejs-code-review/evals.json +74 -0
  159. package/src/gdskills/bundled/stacks/ts-js-node/skills/nodejs-esm-migration/SKILL.md +152 -0
  160. package/src/gdskills/bundled/stacks/ts-js-node/skills/nodejs-esm-migration/evals.json +71 -0
  161. package/src/gdskills/bundled/stacks/ts-js-node/skills/nodejs-implementation/SKILL.md +127 -0
  162. package/src/gdskills/bundled/stacks/ts-js-node/skills/nodejs-implementation/evals.json +72 -0
  163. package/src/gdskills/bundled/stacks/ts-js-node/skills/nodejs-testing/SKILL.md +134 -0
  164. package/src/gdskills/bundled/stacks/ts-js-node/skills/nodejs-testing/evals.json +70 -0
  165. package/src/gdskills/bundled/stacks/vue/agent-refs.json +4 -0
  166. package/src/gdskills/bundled/stacks/vue/governance/eval.json +2215 -0
  167. package/src/gdskills/bundled/stacks/vue/governance/scout.json +42 -0
  168. package/src/gdskills/bundled/stacks/vue/pack.json +42 -0
  169. package/src/gdskills/bundled/stacks/vue/rules/coding-style.mdc +73 -0
  170. package/src/gdskills/bundled/stacks/vue/rules/patterns.mdc +84 -0
  171. package/src/gdskills/bundled/stacks/vue/rules/security.mdc +60 -0
  172. package/src/gdskills/bundled/stacks/vue/rules/testing.mdc +69 -0
  173. package/src/gdskills/bundled/stacks/vue/skills/vue-build-fix/SKILL.md +137 -0
  174. package/src/gdskills/bundled/stacks/vue/skills/vue-build-fix/evals.json +72 -0
  175. package/src/gdskills/bundled/stacks/vue/skills/vue-code-review/SKILL.md +120 -0
  176. package/src/gdskills/bundled/stacks/vue/skills/vue-code-review/evals.json +71 -0
  177. package/src/gdskills/bundled/stacks/vue/skills/vue-implementation/SKILL.md +122 -0
  178. package/src/gdskills/bundled/stacks/vue/skills/vue-implementation/evals.json +72 -0
  179. package/src/gdskills/bundled/stacks/vue/skills/vue-testing/SKILL.md +115 -0
  180. package/src/gdskills/bundled/stacks/vue/skills/vue-testing/evals.json +72 -0
  181. package/src/gdskills/bundled/stacks/vue/skills/vue2-to-vue3-migration/SKILL.md +135 -0
  182. package/src/gdskills/bundled/stacks/vue/skills/vue2-to-vue3-migration/evals.json +71 -0
@@ -0,0 +1,904 @@
1
+ {
2
+ "schemaVersion": "1.0.0",
3
+ "reports": [
4
+ {
5
+ "schemaVersion": "1.0.0",
6
+ "skillId": "mobx/mobx-store-implementation",
7
+ "strictness": "high",
8
+ "trials": 10,
9
+ "triggerAccuracy": {
10
+ "truePositive": 4,
11
+ "falsePositive": 0,
12
+ "positives": 8,
13
+ "negatives": 6
14
+ },
15
+ "evidence": "authored",
16
+ "scenarios": [
17
+ {
18
+ "id": "trigger-positive-1",
19
+ "kind": "trigger-positive",
20
+ "prompt": "Whichever item the user clicks should stay highlighted everywhere it's shown on the page, even after they navigate away and back -- what's the cleanest way to track that in MobX?",
21
+ "strictness": "high",
22
+ "trials": 1,
23
+ "passes": 0,
24
+ "passRate": 0,
25
+ "passAtK": 0,
26
+ "grader": "trigger-rank-fork-family",
27
+ "status": "ran",
28
+ "deterministic": true
29
+ },
30
+ {
31
+ "id": "trigger-positive-2",
32
+ "kind": "trigger-positive",
33
+ "prompt": "I have an async method on this MobX store that hits an API and then updates a few fields -- what's the right way to write that so the state updates correctly?",
34
+ "strictness": "high",
35
+ "trials": 1,
36
+ "passes": 1,
37
+ "passRate": 1,
38
+ "passAtK": 1,
39
+ "grader": "trigger-rank-fork-family",
40
+ "status": "ran",
41
+ "deterministic": true
42
+ },
43
+ {
44
+ "id": "trigger-positive-3",
45
+ "kind": "trigger-positive",
46
+ "prompt": "I've got a new MobX store -- what's the right way to make it available to my React components so they can read from it and re-render when it changes?",
47
+ "strictness": "high",
48
+ "trials": 1,
49
+ "passes": 1,
50
+ "passRate": 1,
51
+ "passAtK": 1,
52
+ "grader": "trigger-rank-fork-family",
53
+ "status": "ran",
54
+ "deterministic": true
55
+ },
56
+ {
57
+ "id": "trigger-positive-4",
58
+ "kind": "trigger-positive",
59
+ "prompt": "This component isn't re-rendering when the store's items array changes, help me fix the wiring",
60
+ "strictness": "high",
61
+ "trials": 1,
62
+ "passes": 1,
63
+ "passRate": 1,
64
+ "passAtK": 1,
65
+ "grader": "trigger-rank-fork-family",
66
+ "status": "ran",
67
+ "deterministic": true
68
+ },
69
+ {
70
+ "id": "trigger-positive-5",
71
+ "kind": "trigger-positive",
72
+ "prompt": "The order total should be derived from the line items in this MobX store rather than kept in sync by hand",
73
+ "strictness": "high",
74
+ "trials": 1,
75
+ "passes": 0,
76
+ "passRate": 0,
77
+ "passAtK": 0,
78
+ "grader": "trigger-rank-fork-family",
79
+ "status": "ran",
80
+ "deterministic": true
81
+ },
82
+ {
83
+ "id": "trigger-positive-6",
84
+ "kind": "trigger-positive",
85
+ "prompt": "This store needs to log something every time a value on it changes, but only while the component using it is actually mounted -- how do I set that up in MobX?",
86
+ "strictness": "high",
87
+ "trials": 1,
88
+ "passes": 0,
89
+ "passRate": 0,
90
+ "passAtK": 0,
91
+ "grader": "trigger-rank-fork-family",
92
+ "status": "ran",
93
+ "deterministic": true
94
+ },
95
+ {
96
+ "id": "trigger-positive-7",
97
+ "kind": "trigger-positive",
98
+ "prompt": "Our settings panel needs its own MobX store that tracks loading and error while it saves",
99
+ "strictness": "high",
100
+ "trials": 1,
101
+ "passes": 0,
102
+ "passRate": 0,
103
+ "passAtK": 0,
104
+ "grader": "trigger-rank-fork-family",
105
+ "status": "ran",
106
+ "deterministic": true
107
+ },
108
+ {
109
+ "id": "trigger-positive-8",
110
+ "kind": "trigger-positive",
111
+ "prompt": "Convert this async store method so the post-await mutations are wrapped correctly",
112
+ "strictness": "high",
113
+ "trials": 1,
114
+ "passes": 1,
115
+ "passRate": 1,
116
+ "passAtK": 1,
117
+ "grader": "trigger-rank-fork-family",
118
+ "status": "ran",
119
+ "deterministic": true
120
+ },
121
+ {
122
+ "id": "trigger-negative-1",
123
+ "kind": "trigger-negative",
124
+ "prompt": "Review this MobX store for accessibility-modifier and method-ordering issues",
125
+ "strictness": "high",
126
+ "trials": 1,
127
+ "passes": 1,
128
+ "passRate": 1,
129
+ "passAtK": 1,
130
+ "grader": "trigger-rank-fork-family",
131
+ "status": "ran",
132
+ "deterministic": true
133
+ },
134
+ {
135
+ "id": "trigger-negative-2",
136
+ "kind": "trigger-negative",
137
+ "prompt": "Write a test that waits for this store's async fetchItems action to resolve before asserting",
138
+ "strictness": "high",
139
+ "trials": 1,
140
+ "passes": 1,
141
+ "passRate": 1,
142
+ "passAtK": 1,
143
+ "grader": "trigger-rank-fork-family",
144
+ "status": "ran",
145
+ "deterministic": true
146
+ },
147
+ {
148
+ "id": "trigger-negative-3",
149
+ "kind": "trigger-negative",
150
+ "prompt": "Add client-side form validation and inline error messages to this plain React form, no state library involved",
151
+ "strictness": "high",
152
+ "trials": 1,
153
+ "passes": 1,
154
+ "passRate": 1,
155
+ "passAtK": 1,
156
+ "grader": "trigger-rank-fork-family",
157
+ "status": "ran",
158
+ "deterministic": true
159
+ },
160
+ {
161
+ "id": "trigger-negative-4",
162
+ "kind": "trigger-negative",
163
+ "prompt": "Fix this tsc error: Property 'total' does not exist on type 'OrderDraft'",
164
+ "strictness": "high",
165
+ "trials": 1,
166
+ "passes": 1,
167
+ "passRate": 1,
168
+ "passAtK": 1,
169
+ "grader": "trigger-rank-fork-family",
170
+ "status": "ran",
171
+ "deterministic": true
172
+ },
173
+ {
174
+ "id": "trigger-negative-5",
175
+ "kind": "trigger-negative",
176
+ "prompt": "Add a new Pinia store action for saving user preferences",
177
+ "strictness": "high",
178
+ "trials": 1,
179
+ "passes": 1,
180
+ "passRate": 1,
181
+ "passAtK": 1,
182
+ "grader": "trigger-rank-fork-family",
183
+ "status": "ran",
184
+ "deterministic": true
185
+ },
186
+ {
187
+ "id": "trigger-negative-6",
188
+ "kind": "trigger-negative",
189
+ "prompt": "Run the project's lint and type-check scripts and report every failure",
190
+ "strictness": "high",
191
+ "trials": 1,
192
+ "passes": 1,
193
+ "passRate": 1,
194
+ "passAtK": 1,
195
+ "grader": "trigger-rank-fork-family",
196
+ "status": "ran",
197
+ "deterministic": true
198
+ },
199
+ {
200
+ "id": "wrap-post-await-mutation",
201
+ "kind": "behavior",
202
+ "prompt": "I have this MobX action:\n\n```ts\nasync fetchItems() {\n this.loading = true;\n const items = await this.service.getAll();\n this.items = items;\n this.loading = false;\n}\n```\n\nWhat needs to change so it works correctly under MobX's enforceActions?",
203
+ "strictness": "high",
204
+ "trials": 10,
205
+ "passes": 10,
206
+ "passRate": 1,
207
+ "passAtK": 1,
208
+ "grader": "judge",
209
+ "status": "ran",
210
+ "trialRecords": [
211
+ {
212
+ "output": "Under `enforceActions: \"always\"` (or `\"observed\"` with observed state), the two assignments after the `await` happen outside an action and will throw. Only the synchronous part before the first `await` runs inside the action's own action context; resuming after the await drops out of it.\n\nThe minimal fix is to wrap every post-`await` mutation in `runInAction`:\n\n```ts\nasync fetchItems() {\n this.loading = true;\n try {\n const items = await this.service.getAll();\n runInAction(() => {\n this.items = items;\n this.loading = false;\n });\n } catch (e) {\n runInAction(() => {\n this.loading = false;\n });\n throw e;\n }\n}\n```\n\nPoints to keep in mind:\n\n- `this.loading = true` before the `await` is fine as-is — that assignment is still inside the action. Only the resumed lines need the wrap.\n- Keep the writes grouped in one `runInAction` block so observers see a single, consistent update rather than an intermediate `items`-set-but-still-loading state.\n- The `catch` matters: without it, a rejected `getAll()` leaves `loading` stuck at `true` forever, since the reset never runs. Reset it in the failure path too.\n- If the store uses the project's async-action pattern, keep the public method thin (guard, then delegate to a `private async` worker that holds the `runInAction` blocks), per `rules/patterns.mdc`.\n- The alternative is MobX `flow`, but for a single action like this `runInAction` is the smaller change and doesn't require converting the method to a generator. Reach for `flow` only when many sequential-await actions make the repeated `runInAction` blocks the dominant noise.\n\nAfter the change, confirm the project's type-check passes and that any test asserting on `fetchItems` still observes `loading` returning to `false` on both the success and rejected paths.",
213
+ "outputSha256": "dd56f9ae961749fbd3dae1afd208234fa8890b71442414e001773afd64561f7e",
214
+ "promptSha256": "6da40c974db51856202177d839563926568dc12999a71d5ebf3c4d02315b4e66",
215
+ "deterministic": [],
216
+ "judge": {
217
+ "verdict": "pass",
218
+ "reason": "Shows corrected code wrapping this.items = items and this.loading = false inside runInAction(() => {...}) after the await (plus a catch block). States the post-await assignments 'happen outside an action' because 'resuming after the await drops out of it.' No enforceActions loosening; no bare post-await assignments remain."
219
+ },
220
+ "passed": true
221
+ },
222
+ {
223
+ "output": "Under `enforceActions: \"always\"` (or `\"observed\"` while the store is observed), the two assignments **after** the `await` happen in a plain async continuation, not inside an action — MobX will throw `[MobX] Since strict-mode is enabled, changing (observed) observable values without using an action is not allowed.`\n\nThe fix is to wrap every post-`await` mutation in `runInAction`:\n\n```ts\nimport { observable, action, runInAction } from \"mobx\";\n\nclass ItemsStore {\n @observable items: Item[] = [];\n @observable loading = false;\n\n constructor(private readonly service: ItemsService) {}\n\n @action.bound\n async fetchItems() {\n this.loading = true;\n\n const items = await this.service.getAll();\n\n runInAction(() => {\n this.items = items;\n this.loading = false;\n });\n }\n}\n```\n\nNotes on why this shape:\n\n- The **first** assignment (`this.loading = true`) is fine as-is — it runs synchronously inside the action before the first `await`. It is only the code that resumes after the `await` that has lost its action context; the action wrapper only covers the synchronous portion of the function.\n- Both post-`await` assignments go in a **single** `runInAction` so the store notifies observers once instead of twice (items, then loading).\n- If `getAll()` throws, the `loading = false` never runs and the spinner sticks. Either wrap the body in `try/finally` with `runInAction` in the `finally`, or catch and surface an error state:\n\n```ts\n @action.bound\n async fetchItems() {\n this.loading = true;\n try {\n const items = await this.service.getAll();\n runInAction(() => {\n this.items = items;\n });\n } finally {\n runInAction(() => {\n this.loading = false;\n });\n }\n }\n```\n\n- The alternative for stores with many sequential-await actions is `flow`, which keeps its generator body in action context across every `yield`, removing the repeated `runInAction` blocks. Only reach for it if that repetition is the dominant noise in the file — for a single action like this one, `runInAction` is the lighter change.\n\nTwo things to check while you're there: `@action.bound` (rather than plain `@action`) if `fetchItems` is passed as a callback or called from the UI without a bound receiver, and confirm the project's `configure(...)` actually enables `enforceActions` — if it's off today, the bug is latent and will surface the moment it's turned on.",
224
+ "outputSha256": "abcb8bf7378fb3da38a771c7a492864517339a0bb661ca89a98eac4b5a17d409",
225
+ "promptSha256": "6da40c974db51856202177d839563926568dc12999a71d5ebf3c4d02315b4e66",
226
+ "deterministic": [],
227
+ "judge": {
228
+ "verdict": "pass",
229
+ "reason": "Shows corrected code wrapping both post-await assignments in runInAction(() => { this.items = items; this.loading = false; }), and states the post-await code runs in a plain async continuation outside the action, losing action context. Also offers flow as alternative. No enforceActions loosening recommended."
230
+ },
231
+ "passed": true
232
+ },
233
+ {
234
+ "output": "Under `enforceActions` (i.e. `configure({ enforceActions: \"always\" })` or `\"observed\"`), every state mutation must happen inside an action. The two assignments after the `await` — `this.items = items` and `this.loading = false` — run in a later microtask, outside the action's synchronous call stack, so they're no longer covered by the action that started the method. At that point the mutation is unattributed, and MobX will either throw (`\"always\"`) or warn (`\"observed\"` when the state is observed).\n\nSo the fix is to wrap the post-`await` mutations in `runInAction`. That re-establishes an action scope around the continuation:\n\n```ts\n@action\nasync fetchItems() {\n this.loading = true;\n const items = await this.service.getAll();\n runInAction(() => {\n this.items = items;\n this.loading = false;\n });\n}\n```\n\nKey points:\n\n- The `this.loading = true` before the `await` is fine as-is — it runs synchronously inside the action.\n- Everything after the `await` must be inside `runInAction`, including the `loading = false` reset. It's easy to remember to wrap the \"real\" data assignment and forget the flag reset, but the flag is a mutation too and will trip the same enforcement.\n- Import `runInAction` from `mobx`.\n- If the method can be called from multiple places and you're using the project's `@action.bound` convention for UI-invoked actions, keep that; the `runInAction` wrapping is orthogonal to whether `action` or `action.bound` is used.\n\nTwo follow-on considerations worth applying while you're in there:\n\n1. Error handling. As written, if `getAll()` rejects, `loading` stays `true` forever. Wrap the awaited call in `try/catch` (or `try/finally`) and reset the flag in the `finally`/catch path, still inside `runInAction`:\n\n```ts\n@action\nasync fetchItems() {\n this.loading = true;\n try {\n const items = await this.service.getAll();\n runInAction(() => {\n this.items = items;\n });\n } finally {\n runInAction(() => {\n this.loading = false;\n });\n }\n}\n```\n\n2. If this store accumulates many sequential-`await` actions, the repeated `runInAction` blocks become the dominant noise. In that case switch the method to a `flow` generator (`fetchItems = flow(function* (this: XyzStore) { ... })`), which keeps every `yield`ed continuation inside an action automatically. For a single `await` like this one, `runInAction` is the simpler and conventional choice — don't reach for `flow` until the repetition is actually warranted.\n\nOne thing to double-check: the store's `configure(...)` call (usually in the app-entry or test-setup file) determines whether a bare post-`await` assignment throws or merely warns, so it may look like it \"works\" in a dev run and still be incorrect. The `runInAction` wrapping is required regardless of which mode you're in.",
235
+ "outputSha256": "2061196e012645e7b5bed2dcfddd7613029edc858b7e601f58cf1eeb2a8646f0",
236
+ "promptSha256": "6da40c974db51856202177d839563926568dc12999a71d5ebf3c4d02315b4e66",
237
+ "deterministic": [],
238
+ "judge": {
239
+ "verdict": "pass",
240
+ "reason": "Shows corrected code wrapping both post-await assignments in runInAction(() => { this.items = items; this.loading = false; }), and states post-await code runs in a later microtask outside the action's synchronous call stack, so mutations are unattributed. Also mentions flow as alternative. No enforceActions loosening recommended."
241
+ },
242
+ "passed": true
243
+ },
244
+ {
245
+ "output": "Your action has two problems under `enforceActions: \"always\"` (or `\"observed\"` with observers active):\n\n1. **The mutations after the `await` are no longer inside an action.** When you `await`, the function suspends and resumes later — by the time `this.items = items` and `this.loading = false` run, you're outside the action's tracking context. MobX will throw (or warn, depending on config) because those are state mutations not wrapped in an action. Everything *before* the `await` (`this.loading = true`) is fine, because it's still synchronous within the action.\n\n2. **The error path leaves `loading` stuck at `true`.** If `this.service.getAll()` rejects, `this.loading = false` never runs.\n\nHere's the corrected version:\n\n```ts\nasync fetchItems() {\n this.loading = true;\n try {\n const items = await this.service.getAll();\n runInAction(() => {\n this.items = items;\n this.loading = false;\n });\n } catch (error) {\n runInAction(() => {\n this.loading = false;\n });\n throw error; // or handle/log as the project expects\n }\n}\n```\n\nKey points:\n\n- **Wrap every post-`await` mutation in `runInAction`.** That re-establishes an action context for the resumed synchronous block, so both `this.items` and `this.loading` assignments are legal.\n- **Add `try`/`catch`** so `loading` is reset on failure too, and decide explicitly whether to swallow, log, or re-throw the error (matching the project's convention).\n- **Keep the action annotated.** `fetchItems` should still carry `@action` (or `@action.bound` if it's invoked from the UI) — the leading `this.loading = true` relies on that, and the method still needs to be an action for the parts that run synchronously.\n\nAn alternative if this store has many sequential-`await` actions and the repeated `runInAction` blocks become noisy: use `flow` from MobX with generator functions (`yield this.service.getAll()`), which keeps mutations action-tracked across awaits automatically. But for a single action like this, `runInAction` is the simpler, more direct fix.\n\nOne thing to verify: check the project's `configure({ enforceActions: ... })` setting. If it's currently `\"never\"` or unset (the default is `\"observed\"`), the bare version might *appear* to work today — but it will break the moment enforcement is turned on, and even without it, the missing `try`/`catch` still leaves `loading` stuck on error. Fix both regardless.",
246
+ "outputSha256": "14179b471c49d274b25ea6bcf556888e152e18d9e042fe12b19cb9ccbab1ff08",
247
+ "promptSha256": "6da40c974db51856202177d839563926568dc12999a71d5ebf3c4d02315b4e66",
248
+ "deterministic": [],
249
+ "judge": {
250
+ "verdict": "pass",
251
+ "reason": "Corrected code wraps both post-await assignments in runInAction(() => { this.items = items; this.loading = false; }) (plus catch block). States mutations after await are 'no longer inside an action' / 'outside the action's tracking context.' Mentions flow as alternative. Does not recommend disabling enforceActions; only warns to verify config."
252
+ },
253
+ "passed": true
254
+ },
255
+ {
256
+ "output": "Under `enforceActions` (whether `\"always\"` or `\"observed\"`), any observable mutation that isn't inside an action throws. The two assignments after the `await` happen in a microtask continuation, outside the original action's synchronous scope, so MobX treats them as mutations from outside an action and throws.\n\n`this.loading = true` before the `await` is fine — that runs synchronously inside the action. The problem is `this.items = items` and `this.loading = false` after it.\n\n## The fix\n\nWrap every post-`await` mutation in `runInAction`:\n\n```ts\nasync fetchItems() {\n this.loading = true;\n try {\n const items = await this.service.getAll();\n runInAction(() => {\n this.items = items;\n this.loading = false;\n });\n } catch (e) {\n runInAction(() => {\n this.loading = false;\n });\n throw e;\n }\n}\n```\n\nEach `runInAction(() => { ... })` opens a fresh action scope around the synchronous block, so the mutations inside are legal.\n\n## What specifically had to change\n\n- `this.items = items` — moved inside `runInAction`.\n- `this.loading = false` — moved inside `runInAction`.\n- `this.loading = true` — unchanged; it's before the `await`, still inside the action.\n- A `try`/`catch` was added so `loading` is also cleared on failure — otherwise a rejected `getAll()` leaves the store stuck in `loading: true` forever.\n\n## The `flow` alternative\n\nIf a store has many actions with this sequential-await shape and the repeated `runInAction` blocks become the dominant noise, use `flow` instead and get automatic action-wrapping of every continuation:\n\n```ts\nfetchItems = flow(function* (this: ItemsStore) {\n this.loading = true;\n try {\n const items = yield this.service.getAll();\n this.items = items;\n this.loading = false;\n } catch (e) {\n this.loading = false;\n throw e;\n }\n});\n```\n\nFor a single action like this one, `runInAction` is the lighter change; reach for `flow` only when it actually reduces noise across the file.\n\n## Notes\n\n- Don't \"fix\" this by turning `enforceActions` off — that silently disables the check that catches exactly this class of bug (mutations that skip observers).\n- If `fetchItems` is invoked from the UI, declare it `@action.bound` (or `makeObservable` annotation `\"action.bound\"`) so `this` is bound and the public entry point is itself an action.\n- After the change, verify `tsc --noEmit` passes and that a test exercises the success and failure paths, including that `loading` returns to `false` on rejection.",
257
+ "outputSha256": "ffd5d70c8e59dbab4aebf3e89872c59855079d04a84eb088f572b2d9cdae9e6c",
258
+ "promptSha256": "6da40c974db51856202177d839563926568dc12999a71d5ebf3c4d02315b4e66",
259
+ "deterministic": [],
260
+ "judge": {
261
+ "verdict": "pass",
262
+ "reason": "Shows corrected code wrapping this.items = items and this.loading = false inside runInAction(() => {...}) after the await, and also gives a flow generator alternative. States post-await code runs in a microtask continuation outside the original action's synchronous scope, so mutations need runInAction. Warns against disabling enforceActions rather than recommending it; no bare post-await assignments remain."
263
+ },
264
+ "passed": true
265
+ },
266
+ {
267
+ "output": "Under `enforceActions: 'always'` (or `'observed'` with observers attached), this action breaks in three places. The `this.loading = true` before the first `await` is fine if the whole method is itself an action — the method's synchronous prologue runs inside the action boundary. But everything after `await this.service.getAll()` runs in a later microtask, *outside* that action boundary, so both `this.items = items` and `this.loading = false` are mutations outside an action and will throw.\n\nTwo additional correctness problems, independent of `enforceActions`:\n\n1. **No error handling.** If `getAll()` rejects, `this.loading` never resets to `false` and the store is stuck loading forever.\n2. **The action isn't `@action.bound`.** If it's passed to the UI as a callback (e.g. `onClick={store.fetchItems}`), `this` will be `undefined` and the mutation-and-await pattern collapses immediately.\n\nSo the fix is: mark it `@action.bound` (or keep `@action` if it's always called as `store.fetchItems()`), keep the synchronous prologue where it is, wrap the post-`await` mutations in `runInAction`, and add a `try/finally` so `loading` always clears.\n\n```ts\n@action.bound\nasync fetchItems(): Promise<void> {\n this.loading = true; // still inside the action boundary\n try {\n const items = await this.service.getAll(); // await exits the action\n runInAction(() => {\n this.items = items; // must be inside runInAction\n this.loading = false; // must be inside runInAction\n });\n } catch (e) {\n runInAction(() => {\n this.loading = false; // must also run on error\n this.error = e; // add an error field if you don't have one\n });\n }\n}\n```\n\nA few notes on the choices:\n\n- **`runInAction` over `flow`.** For a single-await action, `runInAction` is the cleaner fit. Reach for `flow` (generator + `yield`) only when an action has *many* sequential awaits and the repeated `runInAction` blocks dominate the file — see `rules/patterns.mdc`.\n- **Don't widen the action boundary.** You can't simply annotate a `async` method and have MobX treat the post-await code as inside the action; the annotation only covers the synchronous portion. That's precisely why the `runInAction` wrapper is required.\n- **If `enforceActions` is `'observed'` and `loading` is unobserved at the moment of mutation**, the throw won't fire — but the fix is still correct and future-proof, so apply it regardless. Don't rely on the setting being lenient.\n- **`loading`/`error` need to be `@observable`** (and `items` an `@observable.deep` array, or `@observable.ref` if you replace the whole array each fetch). If they aren't observable yet, the `runInAction` wrapper won't help anything re-render — that's the second half of the bridge: observables + actions on the store side, `observer` on the component side.\n- **Verify the wiring.** After this change, confirm the component that calls `fetchItems` is wrapped in `observer` and reads `store.loading`/`store.items` directly (not destructured too early), otherwise you'll fix the throw but still see no re-render.",
268
+ "outputSha256": "58e96cd88b2aa189da1ea67fa5869b542885342b692e4b0597dfa37e07ecfe86",
269
+ "promptSha256": "6da40c974db51856202177d839563926568dc12999a71d5ebf3c4d02315b4e66",
270
+ "deterministic": [],
271
+ "judge": {
272
+ "verdict": "pass",
273
+ "reason": "Corrected code wraps both post-await assignments in runInAction(() => { this.items = items; this.loading = false; }) (also in catch). States post-await code runs in a later microtask outside the action boundary, requiring runInAction. No enforceActions loosening; no bare unwrapped post-await assignments remain."
274
+ },
275
+ "passed": true
276
+ },
277
+ {
278
+ "output": "Under `enforceActions: \"observed\"` (or `\"always\"`), the two assignments after the `await` happen outside any action context, so MobX will throw (under `\"always\"`) or at least warn/throw once the store is observed (under `\"observed\"`). The pre-await `this.loading = true` is fine because `fetchItems` is a decorator'd action, but an `async` function loses its action context across every `await` boundary — the code after the first `await` runs in a fresh microtask with no action wrapping.\n\nThe fix is to wrap the post-`await` mutations in `runInAction`:\n\n```ts\n@action.bound\nasync fetchItems() {\n this.loading = true;\n const items = await this.service.getAll();\n runInAction(() => {\n this.items = items;\n this.loading = false;\n });\n}\n```\n\n`import { runInAction } from \"mobx\";`\n\nA few things worth checking while you're in here:\n\n- **Error path.** The original code has no `try/finally`, so a rejected `getAll()` leaves `this.loading === true` forever. Wrap the whole thing and reset loading in the `finally` (also inside `runInAction`, since resets after the await need it too):\n ```ts\n @action.bound\n async fetchItems() {\n this.loading = true;\n try {\n const items = await this.service.getAll();\n runInAction(() => { this.items = items; });\n } finally {\n runInAction(() => { this.loading = false; });\n }\n }\n ```\n\n- **`flow` alternative.** If the store is dominated by sequential-await actions, `flow` lets you `yield` promises and keeps action context automatically, avoiding the repeated `runInAction` blocks. Only switch if that noise is genuinely the dominant pattern — for a single action like this, `runInAction` is clearer.\n\n- **Async action shape.** Per the skill's `rules/patterns.mdc`, a public async action should stay thin: guard check, then delegate to a `private async` method that does the work. `fetchItems` here is small enough that it's fine inline, but if it grows, split it.\n\n- **Return-early guard.** If calling `fetchItems` twice concurrently is possible, the second call's `loading = false` will clear the flag while the first is still in flight. A `if (this.loading) return;` guard (or tracking a request token) prevents that.\n\n- **Component wiring.** Whatever component renders `items`/`loading` must be wrapped in `observer`, or the (correct) store mutations won't re-render it.\n\nVerify with the project's `tsc --noEmit`, lint, and an existing/new test that awaits `fetchItems()` and asserts both `items` and `loading` settle correctly, including the rejection path if you add the `finally`.",
279
+ "outputSha256": "a191f4151be03b1bc96132848a467c8fd24cf1303073faf5b3372d2c646fa4d2",
280
+ "promptSha256": "6da40c974db51856202177d839563926568dc12999a71d5ebf3c4d02315b4e66",
281
+ "deterministic": [],
282
+ "judge": {
283
+ "verdict": "pass",
284
+ "reason": "Shows corrected code wrapping both post-await assignments in runInAction(() => { this.items = items; this.loading = false; }) (first code block). States code after await runs in a fresh microtask with no action wrapping, losing action context across await. No enforceActions loosening; no bare unwrapped post-await assignments."
285
+ },
286
+ "passed": true
287
+ },
288
+ {
289
+ "output": "Under `enforceActions` (typically `\"observed\"` or `\"always\"`), the synchronous mutations before the first `await` are already fine as long as the method itself is an action — `this.loading = true` runs synchronously inside the action's transaction. The problem is everything after `await this.service.getAll()`:\n\n```ts\nasync fetchItems() {\n this.loading = true; // fine: runs inside the action's sync frame\n const items = await this.service.getAll();\n this.items = items; // NOT in an action anymore\n this.loading = false; // NOT in an action anymore\n}\n```\n\nWhen the `await` resolves, the rest of the function runs in a fresh microtask, **not** inside the original action. `@action`/`action` only wraps the synchronous call frame; it does not survive across `await`. So under `enforceActions` those two assignments either throw (under `\"always\"`, or `\"observed\"` if the fields are observed) or at minimum fail to notify observers correctly.\n\nThere are two correct fixes.\n\n**1. Wrap each post-`await` mutation in `runInAction`** (simplest, matches the `rules/patterns.mdc` async-action shape):\n\n```ts\nasync fetchItems() {\n this.loading = true;\n try {\n const items = await this.service.getAll();\n runInAction(() => {\n this.items = items;\n this.loading = false;\n });\n } catch (error) {\n runInAction(() => {\n this.loading = false;\n // surface the error, e.g. this.error = error\n });\n }\n}\n```\n\nNote the `try/catch`: without it a thrown request leaves `loading` stuck at `true`, and a bare `finally`-less version has the same problem. Prefer catching and setting state over letting the rejection escape.\n\n**2. Convert to a `flow` generator** (the preferred shape when a store has several sequential-`await` actions, so you stop repeating `runInAction` blocks):\n\n```ts\nfetchItems = flow(function* (this: ItemStore) {\n this.loading = true;\n try {\n this.items = yield this.service.getAll();\n this.loading = false;\n } catch (error) {\n this.loading = false;\n }\n});\n```\n\n`flow` runs the whole generator — including every resumed segment after each `yield` — inside the action context, so no `runInAction` is needed. This is the pattern `rules/patterns.mdc` points to for multi-await actions; for a single-await action like this one, option 1 is fine and arguably clearer.\n\n## Other things to line up\n\n- **The method must actually be marked as an action.** If `fetchItems` is a public, UI-invoked method, it should be `@action.bound` (or `action.bound` in the `makeObservable` annotations). If this file uses the public-thin/private-heavy split from the coding style, `fetchItems` stays thin — guard, then delegate to a `private async` method that does the work and holds the `runInAction` blocks.\n- **Add an equality guard** before assigning `this.items` if another store reaction can feed back into this one (`if (items !== this.items)`), per `rules/patterns.mdc`.\n- **Don't rely on `enforceActions` being off.** Even if the project currently configures `configure({ enforceActions: \"never\" })`, the code should still be correct if that setting is tightened later — the `runInAction`/`flow` form is correct either way.\n- **Check the `configure(...)` call** in the app entry or test setup to confirm the project's actual `enforceActions` value; that determines whether the un-wrapped version throws or merely warns, but the fix is the same.\n\n## Verification\n\nAfter the change, confirm: `tsc --noEmit` passes; every post-`await` mutation is inside `runInAction` or the method is a `flow`; a thrown/rejected request leaves `loading === false`; and any existing store test for this action still passes, with a new test covering both the success and rejection paths.",
290
+ "outputSha256": "8855bfb2f01b4b8b632ba0c67c0156929c8bb2d919e957bbc6b6b19a2247a245",
291
+ "promptSha256": "6da40c974db51856202177d839563926568dc12999a71d5ebf3c4d02315b4e66",
292
+ "deterministic": [],
293
+ "judge": {
294
+ "verdict": "pass",
295
+ "reason": "Answer shows corrected code wrapping this.items = items and this.loading = false inside runInAction(() => {...}) (fix 1), and also a flow generator alternative (fix 2). It explicitly states code after await runs in a fresh microtask, not inside the original action, since @action only wraps the sync call frame. No fail criteria: it does not recommend disabling enforceActions (only warns against relying on it being off), and no bare unwrapped post-await assignments remain."
296
+ },
297
+ "passed": true
298
+ },
299
+ {
300
+ "output": "# Fixing this action under `enforceActions`\n\nThe core problem: under `enforceActions: \"always\"` (or the default `\"observed\"` once anything is observing the state), **every** mutation to observable state must happen inside an action. This action mutates `this.loading` and `this.items` **after** the `await`, which lands them outside the synchronous window of the `async fetchItems()` — so MobX throws `[MobX] Since strict-mode is enabled, changing (observed) observable values without using an action is not allowed.`\n\nThe `this.loading = true` at the top is fine (it runs synchronously inside the action). Only the two assignments after the `await` are the problem.\n\n## Option 1: wrap the post-await mutations in `runInAction` (preferred)\n\n```ts\nimport { action, observable, runInAction } from \"mobx\";\n\nclass ItemStore {\n @observable loading = false;\n @observable items: Item[] = [];\n\n constructor(private readonly service: ItemService) {\n makeObservable(this);\n }\n\n @action.bound\n async fetchItems() {\n this.loading = true;\n try {\n const items = await this.service.getAll();\n runInAction(() => {\n this.items = items;\n this.loading = false;\n });\n } catch (err) {\n runInAction(() => {\n this.loading = false;\n });\n // rethrow or store an error field; do not silently swallow\n throw err;\n }\n }\n}\n```\n\nKey points:\n\n- `runInAction` establishes a fresh synchronous action around the assignments, so `enforceActions` is satisfied.\n- `try/catch` with `runInAction` in the `catch` too — otherwise a rejected `getAll()` leaves `loading` stuck on `true` forever.\n- `@action.bound` (or an explicit arrow property) so the method keeps `this` when passed as a callback (e.g. `onClick={store.fetchItems}`). A plain `@action async` method only sees the *caller's* `this` otherwise.\n\n## Option 2: use `flow` (only if this store has many sequential-await actions)\n\n`flow` lets you drop the `runInAction` blocks and keep the async/await style with a generator:\n\n```ts\nimport { flow, observable, makeObservable } from \"mobx\";\n\nfetchItems = flow(function* (this: ItemStore) {\n this.loading = true;\n try {\n const items: Item[] = yield this.service.getAll();\n this.items = items;\n this.loading = false;\n } catch (err) {\n this.loading = false;\n throw err;\n }\n});\n```\n\nEvery `yield` is a suspension point and every resumption is automatically wrapped in an action, so no `runInAction` is needed. Per the skill's guidance, reach for this only when the repeated `runInAction` blocks are genuinely the dominant noise in a store with several sequential-await actions — for one action, Option 1 is clearer.\n\n## What must NOT change\n\n- `this.loading = true` at the top stays as-is — it runs synchronously before the first `await`, still within the action.\n- Don't \"fix\" this by flipping `enforceActions` off. That silences the error but reintroduces the actual bug it's warning about: a mutation that observers may not see. The `runInAction` fix is the correct one.\n\n## Verification\n\nBefore calling this done, confirm:\n\n- The project's type-check (`tsc --noEmit`) exits 0.\n- No observable mutation occurs outside an action — grep the method for assignments after every `await`.\n- A test covers both the success path (`items` set, `loading` back to `false`) and the failure path (`loading` back to `false`, error propagated). See `rules/testing.mdc`.\n- If any component reads `loading`/`items`, it is wrapped in `observer` (from `mobx-react-lite`); without that, the fix may still look broken because nothing re-renders — that's a separate, equally common cause.",
301
+ "outputSha256": "1705894d8f3fb28f213e3d295f3d852d62072e4b57ff2afdb5ff6d5ef19c6c8a",
302
+ "promptSha256": "6da40c974db51856202177d839563926568dc12999a71d5ebf3c4d02315b4e66",
303
+ "deterministic": [],
304
+ "judge": {
305
+ "verdict": "pass",
306
+ "reason": "Option 1 shows corrected code wrapping both post-await assignments (this.items = items; this.loading = false) inside runInAction(() => {...}). It states mutations after await land outside the action's synchronous window, so runInAction re-enters. Option 2 shows flow generator. It explicitly warns against flipping enforceActions off rather than recommending it."
307
+ },
308
+ "passed": true
309
+ },
310
+ {
311
+ "output": "The problem: `enforceActions: \"observed\"` (or `\"always\"`) only allows state mutations inside an action. The two assignments *before* the `await` are fine — they run synchronously inside `fetchItems`, which is an action. The two assignments *after* the `await` are **not** covered: `await` yields, and when the continuation resumes it is no longer running inside the action's synchronous scope. So `this.items = items` and `this.loading = false` will throw (or warn) under `enforceActions`.\n\n## The fix\n\nWrap every post-`await` mutation in `runInAction`:\n\n```ts\nimport { action, makeObservable, observable, runInAction } from \"mobx\";\n\nclass ItemsStore {\n @observable loading = false;\n @observable items: Item[] = [];\n\n constructor(private readonly service: ItemService) {\n makeObservable(this);\n }\n\n @action.bound\n async fetchItems(): Promise<void> {\n this.loading = true; // still in the action — fine as-is\n try {\n const items = await this.service.getAll();\n runInAction(() => { // re-enters an action context\n this.items = items;\n this.loading = false;\n });\n } catch (err) {\n runInAction(() => {\n this.loading = false;\n });\n throw err; // or route to an error field\n }\n }\n}\n```\n\nPoints that matter here:\n\n- **`this.loading = true` needs no change** — it runs synchronously before the `await`, still inside the action.\n- **`runInAction` is required, not stylistic.** It re-establishes the action context for the resumption, so the mutations are legal under `enforceActions` and observers are notified.\n- **Add a `catch`.** The original leaves `loading` stuck `true` on a failed request; resetting it inside `runInAction` in the failure path fixes that. This is also the reason to prefer the explicit `try/catch + runInAction` shape over a single blanket `runInAction` around only the success path.\n- **`@action.bound`** if this action is passed to a component (e.g. `onClick={store.fetchItems}`), so `this` stays bound without a wrapper arrow — match whatever the surrounding store already uses.\n\n## Alternative: `flow`\n\nIf the store has several sequential-await actions, the repeated `runInAction` blocks become the dominant noise in the file. In that case use `flow` instead:\n\n```ts\nimport { flow } from \"mobx\";\n\nfetchItems = flow(function* (this: ItemsStore) {\n this.loading = true;\n try {\n const items: Item[] = yield this.service.getAll();\n this.items = items;\n this.loading = false;\n } catch (err) {\n this.loading = false;\n throw err;\n }\n});\n```\n\n`flow` runs each segment between `yield`s in its own action context, so every mutation is legal without `runInAction`. Reach for it only when that pattern is genuinely recurring — for a single action, the `runInAction` version is clearer.\n\n## Checklist before calling it done\n\n- Every mutation after an `await` sits inside `runInAction` (or the whole action is a `flow` generator).\n- The `catch` path also resets state inside `runInAction`, so `loading` can't get stuck.\n- The project's `configure({ enforceActions: ... })` call (app entry or test setup) is what you tested against — don't assume it's off.\n- The component consuming `loading`/`items` is wrapped in `observer`, or the now-legal mutations still won't re-render it.",
312
+ "outputSha256": "d4ecb0b7a725cfdd49200662d81ad9790d839412850e2895ebd709d5f1f73000",
313
+ "promptSha256": "6da40c974db51856202177d839563926568dc12999a71d5ebf3c4d02315b4e66",
314
+ "deterministic": [],
315
+ "judge": {
316
+ "verdict": "pass",
317
+ "reason": "Answer shows corrected code wrapping this.items = items and this.loading = false inside runInAction(() => {...}) after the await, and states await yields so the continuation is no longer in the action's synchronous scope, requiring runInAction. Also offers flow alternative. No enforceActions loosening; no bare post-await assignments remain."
318
+ },
319
+ "passed": true
320
+ }
321
+ ]
322
+ },
323
+ {
324
+ "id": "component-not-rerendering",
325
+ "kind": "behavior",
326
+ "prompt": "This component reads `store.items.length` in its JSX but never updates when items change:\n\n```tsx\nfunction ItemList({ store }: { store: ItemStore }) {\n return <div>{store.items.length} items</div>;\n}\n```\n\nWhat's wrong and how do I fix it?",
327
+ "strictness": "high",
328
+ "trials": 10,
329
+ "passes": 10,
330
+ "passRate": 1,
331
+ "passAtK": 1,
332
+ "grader": "judge",
333
+ "status": "ran",
334
+ "trialRecords": [
335
+ {
336
+ "output": "The problem is that `ItemList` is a plain function component with no `observer` wrapper. MobX tracks observable reads only inside a reactive context; outside one, `store.items.length` is read exactly once during React's render and nothing subscribes to `items`. When `items` changes, MobX has no observer to notify, so React never re-renders the component.\n\nThere's a second, subtler trap in this exact shape: `store.items.length` reads the observable array **and** the length property. If you'd instead destructured up front, e.g. `const { items } = store;` and then rendered `items.length`, the destructure happens at render time but the `.length` read is still fine here — the real failure is simply the missing subscription. But note that destructuring an observable *object* (`const { count } = store`) or passing `store.someField` into a `useState`/`useMemo` dependency is a common variant that also breaks tracking, so keep reads close to where they're rendered.\n\n## Fix\n\nWrap the component in `observer` from `mobx-react-lite`:\n\n```tsx\nimport { observer } from 'mobx-react-lite';\n\nconst ItemList = observer(function ItemList({ store }: { store: ItemStore }) {\n return <div>{store.items.length} items</div>;\n});\n```\n\nNow the render is a reactive context: MobX records the `items` read, and any action that mutates `items` (add/remove/clear) will schedule a re-render of this component.\n\n## Notes and things to check\n\n- **`items` must actually be observable.** If `ItemStore` declares `items` without `@observable` (or it's a plain array assigned in the constructor and never annotated), `observer` alone won't help. It should be `@observable items: Item[] = []` (or `@observable.ref` if you replace the whole array rather than mutate in place). See `rules/coding-style.mdc`.\n- **Mutations must go through `@action`.** If some code does `store.items.push(...)` outside an action, it either throws under `enforceActions: 'always'` or, worse, mutates without notifying cleanly. Route all mutations through an `@action` on the store.\n- **If `store` is passed as a prop from a non-observer parent, that's fine** — `observer` only needs to wrap the component that reads the observable. But verify the parent isn't re-creating `store` on every render; stores should be stable instances (constructed once, shared via context or a singleton), not rebuilt per render.\n- **Don't reach for `key` remounts or `forceUpdate`.** Those paper over the missing subscription and reintroduce the bug the moment the workaround is removed.\n\n## Verification\n\n- Add `observer` (and confirm the `@observable`/`@action` annotations on the store).\n- Type-check passes (`tsc --noEmit`).\n- A test or manual check: dispatch the store action that adds an item and confirm `ItemList` re-renders with the new count. With a React Testing Library test, `act(() => store.addItem(...))` followed by an assertion on the new text is the cleanest proof.\n\nThat single change — adding `observer` — fixes the reported symptom, provided the store fields are properly observable and mutations are actions.",
337
+ "outputSha256": "0ac4b31db7f777cba97350941b664f03a17840a98a550ce79199e28c9a66f204",
338
+ "promptSha256": "1aaf3617fa24c9a51593696c18acd0291e399675f1c323b03ca4f825bc57de62",
339
+ "deterministic": [],
340
+ "judge": {
341
+ "verdict": "pass",
342
+ "reason": "Answer states ItemList is a plain component with no observer wrapper, so store.items.length is read once and nothing subscribes, causing no re-render (criterion 1). It shows corrected code: `const ItemList = observer(function ItemList({ store }: { store: ItemStore }) { return <div>{store.items.length} items</div>; });` (criterion 2). It explicitly warns against key remounts/forceUpdate, not recommending them, so no fail criterion holds."
343
+ },
344
+ "passed": true
345
+ },
346
+ {
347
+ "output": "The problem isn't the read itself — `store.items.length` is a fine thing to render. The problem is that `ItemList` is a plain function component, not wrapped in `observer`. MobX tracks observable reads during render, but only if the component is an `observer` component; without that wrapper, React just renders once and never subscribes to `items`, so a mutation to `store.items` notifies nobody and `ItemList` never re-renders.\n\n## Fix\n\nWrap the component in `observer` from `mobx-react-lite`:\n\n```tsx\nimport { observer } from 'mobx-react-lite';\n\nconst ItemList = observer(function ItemList({ store }: { store: ItemStore }) {\n return <div>{store.items.length} items</div>;\n});\n```\n\nKeep the `store` prop as-is (or source it from a context hook — see below). The `observer` wrapper is what registers the observable dependency on `items` and subscribes the component for re-render.\n\n## Two things to verify while you're here\n\n1. **`items` is actually observable.** If the store declares it as a plain array field without `@observable` (or without being picked up by `makeAutoObservable`), wrapping in `observer` still won't help. Check `rules/coding-style.mdc` for the observable-collection typing the project uses — this is the second most common cause right after the missing wrapper.\n\n2. **The `length` read happens during render.** In your snippet it does, so you're fine. This matters because MobX's dependency tracking is per-property-read: if a value is read somewhere outside the reactive render path (e.g. pulled into a variable in an enclosing scope, or destructured early and passed down), the component that actually renders the value may not be the one that subscribed. Here `store.items.length` is read inline in JSX, which is exactly right.\n\n## If the project uses a context hook instead of a prop\n\nSome codebases expose the store via a typed context plus `useXyzStore()` rather than passing it as a prop. If that's this project's convention, the fix is the same `observer` wrap, but the store comes from the hook:\n\n```tsx\nconst ItemList = observer(function ItemList() {\n const store = useItemStore();\n return <div>{store.items.length} items</div>;\n});\n```\n\nMatch whichever wiring the surrounding stores already use — do not introduce a second style into the codebase.\n\n## What not to do\n\nDon't reach for a `key` prop or a forced remount to \"fix\" this. That works around the symptom (component eventually shows fresh data on an unrelated re-render trigger) without subscribing it, so it'll silently lag again the next time items change with no other re-render cause. Wrap in `observer` and the subscription is real.",
348
+ "outputSha256": "411faaf1ca1fc462f2c7f6f1141766acf75da2f975bad4034e16365476699910",
349
+ "promptSha256": "1aaf3617fa24c9a51593696c18acd0291e399675f1c323b03ca4f825bc57de62",
350
+ "deterministic": [],
351
+ "judge": {
352
+ "verdict": "pass",
353
+ "reason": "Answer states ItemList is a plain function component not wrapped in observer, so React renders once and never subscribes to items (criterion 1). It shows corrected code: `const ItemList = observer(function ItemList({ store }: { store: ItemStore }) { return <div>{store.items.length} items</div>; });` (criterion 2). It warns against key/forced remount rather than recommending it, so no fail criterion holds."
354
+ },
355
+ "passed": true
356
+ },
357
+ {
358
+ "output": "The component isn't wrapped in `observer`, so MobX never sets up a reaction that re-renders it when `store.items` changes. It reads `store.items.length` on the initial render and then goes silent — the render is untracked.\n\nFix: wrap the component in `observer` from `mobx-react-lite`.\n\n```tsx\nimport { observer } from 'mobx-react-lite';\n\nconst ItemList = observer(function ItemList({ store }: { store: ItemStore }) {\n return <div>{store.items.length} items</div>;\n});\n```\n\nAfter that, the read of `store.items.length` inside the tracked render function registers a dependency, and any action that mutates `items` (push, splice, replace, etc.) re-renders it.\n\nKey points:\n\n- Make sure `items` is actually observable on the store (`@observable` or `@observable.ref`/`.shallow`, depending on whether you replace the array or mutate it in place). If it's a plain field, `observer` alone won't help. If you replace the array, either use `@observable` (deep) and assign a new array inside an `@action`, or use `@observable.ref` so reference changes are the trigger.\n- The mutation must go through an `@action` (or `runInAction`), especially after an `await`. An out-of-action mutation may throw under `enforceActions`, or update the value without notifying observers.\n- Keep the read inside the render body (as it is here). Don't hoist it out of the component's reactive scope in a way that breaks tracking — reading `store.items.length` directly in JSX is exactly right.\n\nSo the two-line fix is the `observer` wrapper plus confirming `items` is observable and mutated in an action. If it still doesn't update after that, check whether the store ever replaces `items` with a new array outside an action, or whether anything else in the tree is caching the component.",
359
+ "outputSha256": "6742ab59348bab50f333843200637f8d12eb4e837853b8b990d6405ae2eeda6e",
360
+ "promptSha256": "1aaf3617fa24c9a51593696c18acd0291e399675f1c323b03ca4f825bc57de62",
361
+ "deterministic": [],
362
+ "judge": {
363
+ "verdict": "pass",
364
+ "reason": "Answer states the component isn't wrapped in observer so MobX never sets up a reaction, reading store.items.length once then going silent (criterion 1). It shows corrected code wrapping ItemList itself: `const ItemList = observer(function ItemList({ store }: { store: ItemStore }) { return <div>{store.items.length} items</div>; });` (criterion 2). No key/forceUpdate/useState workaround; no parent-only wrapping."
365
+ },
366
+ "passed": true
367
+ },
368
+ {
369
+ "output": "The component is missing the `observer` wrapper. `store.items.length` is read inside JSX, which is plain React rendering — React has no idea that `items` is a MobX observable, so it only re-renders when its own state/props change. MobX's reactivity only kicks in for components wrapped in `observer` (from `mobx-react-lite`), which creates a tracking reaction around the render and re-runs it when any observable read during that render changes.\n\nFix:\n\n```tsx\nimport { observer } from 'mobx-react-lite';\n\nconst ItemList = observer(function ItemList({ store }: { store: ItemStore }) {\n return <div>{store.items.length} items</div>;\n});\n```\n\nIf the component is exported directly, apply `observer` at the export site:\n\n```tsx\nexport const ItemList = observer(ItemListBase);\n```\n\nA few things to check while you're here, since a missing `observer` is the most common cause but not the only one:\n\n- **Read too early / destructured away from use.** `observer` tracks observables read *during render*. If someone writes `const { items } = store;` at the top and then renders `items.length`, that read still happens during render, so it's fine — but if a value is hoisted into a variable outside the reactive read path (e.g., captured in a `useMemo` dependency that isn't observable, or passed through a non-observer parent that re-renders on its own schedule), tracking breaks. Keep the observable read where it's rendered.\n- **The parent isn't wrapping this component correctly.** If `ItemList` is rendered inside another component that reads the same store but *that* component isn't an `observer`, the parent may not re-render, so `ItemList` never gets a chance to. Wrap every component in the tree that reads this store in `observer`.\n- **The mutation isn't going through an action.** If `items` is mutated outside an `@action` (e.g., a bare `store.items.push(...)` after an `await` with `enforceActions` off), observers may not be notified. Check that every mutation is inside an action or wrapped in `runInAction` — otherwise even a correctly wrapped `observer` component won't update.\n\nSo: wrap the component in `observer`, and confirm the mutation side is going through a proper action.",
370
+ "outputSha256": "9c13864f233f61d81904a9e7e8f64edfa621a3e4360a080d68d1926a839e3ad3",
371
+ "promptSha256": "1aaf3617fa24c9a51593696c18acd0291e399675f1c323b03ca4f825bc57de62",
372
+ "deterministic": [],
373
+ "judge": {
374
+ "verdict": "pass",
375
+ "reason": "Answer states ItemList is missing the observer wrapper from mobx-react-lite and that this is why it only re-renders on state/props change, not observable changes. It shows corrected code: `const ItemList = observer(function ItemList({ store }: { store: ItemStore }) => ...)`. No fail criteria: it does not recommend key/forceUpdate/useState mirroring, and it wraps ItemList itself, not just a parent."
376
+ },
377
+ "passed": true
378
+ },
379
+ {
380
+ "output": "The component isn't wrapped in `observer`. `store.items.length` is read inside JSX, but MobX only tracks that read and re-runs the component on change if the component is an observer. A plain function component renders once and then never subscribes to the observable, so mutations to `items` don't trigger a re-render.\n\nFix by wrapping it in `observer` from `mobx-react-lite`:\n\n```tsx\nimport { observer } from \"mobx-react-lite\";\n\nconst ItemList = observer(function ItemList({ store }: { store: ItemStore }) {\n return <div>{store.items.length} items</div>;\n});\n```\n\nKeep the read (`store.items.length`) inside the render body, as it already is. If you hoisted it out — e.g. destructuring `const { items } = store` far above, or passing `store.items.length` in as a prop from an unobserved parent — you'd move the read outside the tracked render and break fine-grained tracking even with `observer` present. Here the read is in the JSX, so wrapping the component is sufficient.\n\nTwo things that are not the fix, even though they sometimes appear to work:\n\n- Adding a `key` or forcing a remount: that masks the missing subscription and re-breaks on the next change.\n- Wrapping the parent instead: `observer` must be on the component that actually reads the observable, not only on an ancestor.\n\nAfter the change, the store side needs nothing new — `items` just has to be an `@observable` (or `observable.array`) field and mutations to it must happen inside an `@action` / `runInAction` so observers are notified under `enforceActions`. If the list also doesn't update after mutations that originate from an async method, check that those post-`await` mutations are wrapped in `runInAction` — a missing `observer` and an unwrapped async mutation are the two independent causes of \"the UI is stale,\" and this component has the first one.",
381
+ "outputSha256": "323e75d00eb52e2f0e714ac36e7ab2ab9f33e0435cb61f30d910e03f65048811",
382
+ "promptSha256": "1aaf3617fa24c9a51593696c18acd0291e399675f1c323b03ca4f825bc57de62",
383
+ "deterministic": [],
384
+ "judge": {
385
+ "verdict": "pass",
386
+ "reason": "Answer states ItemList isn't wrapped in observer, explains plain component renders once and never subscribes so mutations don't re-render (criterion 1). Shows corrected code: `const ItemList = observer(function ItemList({ store }: { store: ItemStore }) { return <div>{store.items.length} items</div>; });` (criterion 2). It warns against key/remount and parent-only wrapping rather than recommending them, so no fail criteria hold."
387
+ },
388
+ "passed": true
389
+ },
390
+ {
391
+ "output": "## What's wrong\n\nThe component reads `store.items.length` directly in JSX, which is exactly the kind of read MobX's reactivity is designed to track — but it will never re-render on change for one reason: **`ItemList` is not wrapped in `observer`.**\n\nA plain function component is not a MobX reaction. MobX only re-runs a component's render when something has subscribed that component's render function to the observables it read. That subscription is what `observer()` installs. Without it, MobX updates `items` internally, the observable's bookkeeping is correct, and React simply never hears about it — so the component renders once with whatever `items.length` was at first mount and then goes stale forever.\n\nThis is the single most common \"MobX isn't working\" symptom, and it's worth ruling out before anything else (before suspecting `makeObservable` annotations, before adding `key` props, before forcing remounts).\n\nThere is a second, quieter failure mode to check too: even *with* `observer`, a component stops tracking a value if it reads the observable too early and only renders the extracted primitive. If someone had written `const count = store.items.length;` at the top of the component and then rendered `count`, tracking still works *in this case* (the read is inside the tracked render), but the pattern becomes fragile as soon as the read moves into a `useMemo`/`useState` initializer or gets destructured in a parent and passed down as a prop. Here, though, the read is genuinely inside the JSX, so the only defect is the missing wrapper.\n\n## The fix\n\nWrap the component in `observer` from `mobx-react-lite`:\n\n```tsx\nimport { observer } from 'mobx-react-lite';\n\nconst ItemList = observer(function ItemList({ store }: { store: ItemStore }) {\n return <div>{store.items.length} items</div>;\n});\n```\n\nThat's the whole fix for this component. The read of `store.items.length` inside the tracked render now subscribes the component to `store.items`, and MobX will schedule a re-render whenever the collection changes.\n\nA few things to get right while you're here, in order of how likely they are to bite next:\n\n**Use `observer` consistently.** Every component that reads observable state needs it. A missing `observer` on one component in a tree is easy to miss because the parent may re-render for unrelated reasons and mask it — until the parent stops re-rendering, and then the child silently freezes.\n\n**Keep the observable read in the render path.** Don't hoist `store.items.length` into a `useMemo`, a `useState` initializer, or a ref. MobX tracks reads that happen during the tracked render; a read that happens inside a memo with stale dependencies or only on first mount won't re-subscribe. If a component only needs *one* derived value, a `@computed get count()` on the store is the cleaner pattern — it also avoids recomputing the derivation on every render and keeps the derivation consistent across every component that needs it.\n\n```ts\n// in the store\n@computed get itemCount() { return this.items.length; }\n```\n\nThen the component reads `store.itemCount`, and the derivation lives in the store's reactivity graph where it belongs.\n\n**Make sure the store is actually observable.** If `ItemStore` was constructed without `makeObservable`/`makeAutoObservable` (or the `items` field isn't annotated), `items` is a plain array and no amount of `observer` will help — the component would re-render only when the *reference* changes, not on `push`/`splice`. Worth confirming in the store's constructor:\n\n```ts\nconstructor() {\n makeObservable(this, {\n items: observable,\n itemCount: computed,\n addItem: action.bound,\n });\n}\n```\n\n**Don't work around it by remounting.** If you catch yourself reaching for a `key` prop change, a `forceUpdate`, a `useState` counter bumped from a subscription, or an external `autorun` that calls `setState` — stop. Those all suppress the symptom while leaving the component untracked, and they'll resurface as soon as the workaround's trigger condition changes. The fix is the `observer` wrapper (plus an observable store), not a remount.\n\n## Verify\n\nAfter adding `observer`, confirm:\n\n- The project's type-check passes.\n- `ItemList` now re-renders when `store.items` is mutated through a store action (a quick test that pushes to `items` inside `runInAction` and asserts the rendered output is the reliable check).\n- No other component in the same feature reads store state without `observer` — scan the feature's components, not just this one, since the same omission usually happens in clusters.\n- If you switched to a `@computed get itemCount()`, that getter is annotated as `computed` and contains no mutations.",
392
+ "outputSha256": "16c1d093daa5568ca283d99cb65377a97a97247d099f6627c8f41dc79e52fb71",
393
+ "promptSha256": "1aaf3617fa24c9a51593696c18acd0291e399675f1c323b03ca4f825bc57de62",
394
+ "deterministic": [],
395
+ "judge": {
396
+ "verdict": "pass",
397
+ "reason": "Answer states ItemList is not wrapped in observer and that's why it renders once and goes stale ('renders once with whatever items.length was at first mount and then goes stale forever'). It shows corrected code wrapping ItemList itself: `const ItemList = observer(function ItemList({ store }: { store: ItemStore }) {...})`. It explicitly warns against key props/forceUpdate/useState workarounds rather than recommending them."
398
+ },
399
+ "passed": true
400
+ },
401
+ {
402
+ "output": "The component isn't wrapped in `observer`, so it has no reactive subscription to the observable at all. React re-renders it only when its props/state change — and `store` is a stable reference, so changing `store.items` never causes a re-render.\n\n**Fix — wrap the component in `observer`** (from `mobx-react-lite`):\n\n```tsx\nimport { observer } from 'mobx-react-lite';\n\nconst ItemList = observer(function ItemList({ store }: { store: ItemStore }) {\n return <div>{store.items.length} items</div>;\n});\n```\n\n`observer` makes the render function track whatever observables it reads (here `store.items` and its `.length`) and re-render when those change. Reading `store.items.length` *inside* the JSX is exactly right — the read happens during the tracked render, so fine-grained tracking works.\n\nA few things to check alongside this:\n\n- **Don't destructure the observable value before render.** `const { items } = store;` outside the JSX (or at the top of a component that isn't the render body) can break tracking if the read escapes the observed render. Keep the read inside the JSX or inside the `observer` body.\n- **If `ItemList` receives `store` from a parent that also renders it**, still wrap `ItemList` in `observer` — the parent being observed doesn't make the child reactive.\n- **If you're on `mobx-react` (not `-lite`)**, the same `observer` HOC applies; just import from the package the project already uses.\n- **Confirm `items` is actually observable** (`@observable items = []` or equivalent). `observer` can't help with a plain non-observable array field — if it's not observable, make it so per the project's store conventions (`makeObservable` annotations or `makeAutoObservable`).\n\nOnce wrapped, editing the store's `items` from an action will notify this component and it will re-render.\n\nQuick verification: the type-check passes, the component is wrapped in `observer`, and a test that mutates `store.items` and asserts the rendered text updates goes from red (before the fix) to green.",
403
+ "outputSha256": "383d4571692a8df382258aa73707f5e40c0d4bc274f5dd460f3fbb9d02585f5e",
404
+ "promptSha256": "1aaf3617fa24c9a51593696c18acd0291e399675f1c323b03ca4f825bc57de62",
405
+ "deterministic": [],
406
+ "judge": {
407
+ "verdict": "pass",
408
+ "reason": "Answer states ItemList isn't wrapped in observer so it has no reactive subscription and store is a stable reference, so store.items changes never re-render. It shows corrected code: `const ItemList = observer(function ItemList({ store }: { store: ItemStore }) { return <div>{store.items.length} items</div>; });` importing observer from mobx-react-lite. No key/forceUpdate/useState workaround; explicitly warns parent observation doesn't make child reactive."
409
+ },
410
+ "passed": true
411
+ },
412
+ {
413
+ "output": "The component is missing the `observer` wrapper. MobX only tracks reads and re-renders a component if that component is wrapped in `observer` (from `mobx-react-lite`). As written, `ItemList` is a plain function component: React has no subscription to `store.items`, so it renders once and never again when `items` changes. It also won't re-render on prop changes from a parent that isn't itself observing the store.\n\nFix: wrap it in `observer`.\n\n```tsx\nimport { observer } from 'mobx-react-lite';\n\nconst ItemList = observer(function ItemList({ store }: { store: ItemStore }) {\n return <div>{store.items.length} items</div>;\n});\n```\n\nNotes and things to check:\n\n- Read the observable *inside* the `observer` render, as above. Fine-grained tracking only works for reads that happen during render. If you destructure too early or hoist the read into a non-observed helper, you can still lose reactivity:\n ```tsx\n // Bad: read happens before/outside observer tracking of this component's render\n const { items } = store;\n return <div>{items.length} items</div>;\n ```\n Keeping `store.items.length` inline (or via a `@computed` on the store) is safest.\n\n- `store.items.length` on an observable array is tracked correctly, so no change needed there. If `items` were a plain array behind a `.ref`/non-observable field, `observer` alone wouldn't help — the array itself must be `observable` (or the field `@observable`), so confirm `items` is an observable array on `ItemStore`.\n\n- If this component is nested under a parent that receives `store` from a context hook, the parent doesn't need `observer` for this fix to work — but any other component that also reads store state will need its own `observer` wrapper. A missing `observer` on the leaf is the most common cause of exactly this \"reads store but never updates\" symptom.\n\n- Avoid the temptation to force a remount with a changing `key` or a local `useState` bump. That hides the missing subscription instead of fixing it and will still be wrong the next time the store changes in a way React doesn't notice.\n\nAfter the change: type-check passes, and `ItemList` re-renders when `store.items` is mutated through an action (e.g. `store.addItem(...)`). If you have a test for the store, add one that asserts the component updates after an action so this doesn't silently regress.",
414
+ "outputSha256": "2fffd7b4bf2f5b78e4a3360fcd7b6760c650c162554eb8c603aeb93a50c74e28",
415
+ "promptSha256": "1aaf3617fa24c9a51593696c18acd0291e399675f1c323b03ca4f825bc57de62",
416
+ "deterministic": [],
417
+ "judge": {
418
+ "verdict": "pass",
419
+ "reason": "Answer states ItemList is missing the observer wrapper from mobx-react-lite, explaining it renders once and never again when items change. It shows corrected code: `const ItemList = observer(function ItemList({ store }: { store: ItemStore }) { return <div>{store.items.length} items</div>; });`. It warns against key/useState bump without recommending it, and does not wrap only a parent."
420
+ },
421
+ "passed": true
422
+ },
423
+ {
424
+ "output": "# The bug: `ItemList` is not wrapped in `observer`\n\n`store.items.length` is only tracked by MobX if the component reading it is a reactive observer. A plain function component has no subscription to the observable graph, so when `store.items` changes, MobX has no reason to notify this component and React has no reason to re-render it. The value renders once with whatever length it had on mount, then goes stale forever.\n\nThe fix is to wrap the component with `observer` from `mobx-react-lite`:\n\n```tsx\nimport { observer } from 'mobx-react-lite';\n\nconst ItemList = observer(function ItemList({ store }: { store: ItemStore }) {\n return <div>{store.items.length} items</div>;\n});\n```\n\nNotes on the fix:\n\n- Wrap the whole component, not just the returned JSX. `observer` works by wrapping the render function so it can track which observables that render reads; wrapping the JSX expression does nothing.\n- The read of `store.items.length` must happen *inside* the observer's render function. Reading it out in a parent and passing the number down defeats tracking, because by the time `ItemList` sees the prop it's already a plain number. If you need to pass data down, pass the observable itself (or read it in the child that is itself wrapped in `observer`).\n- Don't reach for a `key` prop, a forced remount, or manual subscription as a fix — those are the red-flag workarounds in this skill. A missing `observer` wrapper is the first thing to check for \"component doesn't re-render on store change,\" and it's almost always the answer.\n\nIf this component receives `store` via props, that's fine, but keep an eye on the project's convention in Step 5 of the workflow: if the codebase exposes stores through a typed context + `useXyzStore()` hook, prefer that over prop-drilling the store, and make sure the hook itself throws when the context is unset rather than silently returning undefined.\n\nTo verify: after adding `observer`, changing `store.items` (via the store's action) should re-render this component with the new count. If it still doesn't, the next suspect is an observable that isn't actually observable (missing `@observable` on `items`, or a plain array assigned outside an action under `enforceActions`), not the component.",
425
+ "outputSha256": "69c8b567db2e5b3708e7968f05761ef1ca4cc9acf4834bb6f5e2480efa68a400",
426
+ "promptSha256": "1aaf3617fa24c9a51593696c18acd0291e399675f1c323b03ca4f825bc57de62",
427
+ "deterministic": [],
428
+ "judge": {
429
+ "verdict": "pass",
430
+ "reason": "Answer states ItemList lacks observer wrapper from mobx-react-lite, explaining it renders once then goes stale ('renders once... then goes stale forever'). Shows corrected code: `const ItemList = observer(function ItemList({ store }: { store: ItemStore }) => ...)`. Also warns against key prop/forced remount as workarounds, not recommending them."
431
+ },
432
+ "passed": true
433
+ },
434
+ {
435
+ "output": "The component isn't wrapped in `observer`, so it never subscribes to the `items` observable — React renders it once and MobX has no way to tell it to re-render when the array changes.\n\nFix:\n\n```tsx\nimport { observer } from 'mobx-react-lite';\n\nconst ItemList = observer(function ItemList({ store }: { store: ItemStore }) {\n return <div>{store.items.length} items</div>;\n});\n```\n\n`observer` wraps the component so that any observable it reads during render — here `store.items` (and its `.length`) — is tracked, and the component re-renders when that observable changes.\n\nTwo things to watch when doing the fix:\n\n- Read the observable **inside** the observer's render, not before. If someone later hoists it, e.g. `const { items } = store;` in a parent and passes `items` down as a plain prop, tracking breaks again — pass the store (or a getter) down, or wrap the child in `observer` too.\n- Keep using the store instance from props/context; don't destructure a snapshot of observable data into local variables and expect reactivity.\n\nBefore assuming it's fixed, verify:\n\n- The project's type-check passes (`tsc --noEmit`).\n- `ItemList` (and any other component reading store state) is wrapped in `observer`, per the skill's Step 5.\n- A quick test or manual check: mutate `store.items` inside a `@action` and confirm the count re-renders.\n\nOne caveat: if after adding `observer` it *still* doesn't update, the next thing to check is the mutation path — if items are being assigned after an `await` outside an action, the mutation may not be notifying observers at all. That belongs in the store, not the component, and should be wrapped in `runInAction`.",
436
+ "outputSha256": "c0890120df8052aa37346dc41c1cb8b1d52c35ebdc52bc4bcad11196e8e5197c",
437
+ "promptSha256": "1aaf3617fa24c9a51593696c18acd0291e399675f1c323b03ca4f825bc57de62",
438
+ "deterministic": [],
439
+ "judge": {
440
+ "verdict": "pass",
441
+ "reason": "Answer states ItemList isn't wrapped in observer so it never subscribes and renders once (criterion 1), and shows corrected code `const ItemList = observer(function ItemList({ store }: { store: ItemStore }) { return <div>{store.items.length} items</div>; });` (criterion 2). No key/forceUpdate/useState workaround; no parent-only wrapping claim."
442
+ },
443
+ "passed": true
444
+ }
445
+ ]
446
+ }
447
+ ],
448
+ "verdict": "fail",
449
+ "scope": "bundled",
450
+ "skillDigest": "76d2c5b465869c529e577e1522d51e62e4c65452cdde48fe7195aa241c14246d",
451
+ "catalogDigest": "fd9b6a086f61a996f761a68f9e7a58cde1ce62e121f276d70bbef8e05e372f4b",
452
+ "judgePromptVersion": "2026-09-25.1",
453
+ "runner": "deepseek",
454
+ "model": "deepseek-chat",
455
+ "runnerPromptVersion": "2026-09-25.1",
456
+ "recordedAt": "2026-09-25T15:57:58.326Z",
457
+ "judge": "deepseek",
458
+ "judgeModel": "deepseek-chat"
459
+ },
460
+ {
461
+ "schemaVersion": "1.0.0",
462
+ "skillId": "mobx/mobx-observable-testing",
463
+ "strictness": "high",
464
+ "trials": 10,
465
+ "triggerAccuracy": {
466
+ "truePositive": 7,
467
+ "falsePositive": 0,
468
+ "positives": 7,
469
+ "negatives": 6
470
+ },
471
+ "evidence": "authored",
472
+ "scenarios": [
473
+ {
474
+ "id": "trigger-positive-1",
475
+ "kind": "trigger-positive",
476
+ "prompt": "I need a test for this MobX store's async method that fetches data from the API -- how do I make sure it actually waits for the result before I assert?",
477
+ "strictness": "high",
478
+ "trials": 1,
479
+ "passes": 1,
480
+ "passRate": 1,
481
+ "passAtK": 1,
482
+ "grader": "trigger-rank-fork-family",
483
+ "status": "ran",
484
+ "deterministic": true
485
+ },
486
+ {
487
+ "id": "trigger-positive-2",
488
+ "kind": "trigger-positive",
489
+ "prompt": "My store test asserts on this.items right after calling the async action and it's flaky",
490
+ "strictness": "high",
491
+ "trials": 1,
492
+ "passes": 1,
493
+ "passRate": 1,
494
+ "passAtK": 1,
495
+ "grader": "trigger-rank-fork-family",
496
+ "status": "ran",
497
+ "deterministic": true
498
+ },
499
+ {
500
+ "id": "trigger-positive-3",
501
+ "kind": "trigger-positive",
502
+ "prompt": "How do I test that this MobX store actually does something automatically when a value changes?",
503
+ "strictness": "high",
504
+ "trials": 1,
505
+ "passes": 1,
506
+ "passRate": 1,
507
+ "passAtK": 1,
508
+ "grader": "trigger-rank-fork-family",
509
+ "status": "ran",
510
+ "deterministic": true
511
+ },
512
+ {
513
+ "id": "trigger-positive-4",
514
+ "kind": "trigger-positive",
515
+ "prompt": "Write a test proving dispose() stops this store's reaction from running again",
516
+ "strictness": "high",
517
+ "trials": 1,
518
+ "passes": 1,
519
+ "passRate": 1,
520
+ "passAtK": 1,
521
+ "grader": "trigger-rank-fork-family",
522
+ "status": "ran",
523
+ "deterministic": true
524
+ },
525
+ {
526
+ "id": "trigger-positive-5",
527
+ "kind": "trigger-positive",
528
+ "prompt": "Should enforceActions be relaxed in my MobX store test setup?",
529
+ "strictness": "high",
530
+ "trials": 1,
531
+ "passes": 1,
532
+ "passRate": 1,
533
+ "passAtK": 1,
534
+ "grader": "trigger-rank-fork-family",
535
+ "status": "ran",
536
+ "deterministic": true
537
+ },
538
+ {
539
+ "id": "trigger-positive-6",
540
+ "kind": "trigger-positive",
541
+ "prompt": "How do I test that a derived value on this MobX store updates correctly when the thing it depends on changes?",
542
+ "strictness": "high",
543
+ "trials": 1,
544
+ "passes": 1,
545
+ "passRate": 1,
546
+ "passAtK": 1,
547
+ "grader": "trigger-rank-fork-family",
548
+ "status": "ran",
549
+ "deterministic": true
550
+ },
551
+ {
552
+ "id": "trigger-positive-7",
553
+ "kind": "trigger-positive",
554
+ "prompt": "This store test sometimes reads stale state before the reaction has run, fix the wait",
555
+ "strictness": "high",
556
+ "trials": 1,
557
+ "passes": 1,
558
+ "passRate": 1,
559
+ "passAtK": 1,
560
+ "grader": "trigger-rank-fork-family",
561
+ "status": "ran",
562
+ "deterministic": true
563
+ },
564
+ {
565
+ "id": "trigger-negative-1",
566
+ "kind": "trigger-negative",
567
+ "prompt": "Add a new @action.bound method to this store's constructor that fetches items from the API and wraps the loading/error update in runInAction",
568
+ "strictness": "high",
569
+ "trials": 1,
570
+ "passes": 1,
571
+ "passRate": 1,
572
+ "passAtK": 1,
573
+ "grader": "trigger-rank-fork-family",
574
+ "status": "ran",
575
+ "deterministic": true
576
+ },
577
+ {
578
+ "id": "trigger-negative-2",
579
+ "kind": "trigger-negative",
580
+ "prompt": "Write a React Testing Library test that clicks this button and checks the handler prop was called",
581
+ "strictness": "high",
582
+ "trials": 1,
583
+ "passes": 1,
584
+ "passRate": 1,
585
+ "passAtK": 1,
586
+ "grader": "trigger-rank-fork-family",
587
+ "status": "ran",
588
+ "deterministic": true
589
+ },
590
+ {
591
+ "id": "trigger-negative-3",
592
+ "kind": "trigger-negative",
593
+ "prompt": "Add a @computed get totalPrice getter to this store class that sums the items array",
594
+ "strictness": "high",
595
+ "trials": 1,
596
+ "passes": 1,
597
+ "passRate": 1,
598
+ "passAtK": 1,
599
+ "grader": "trigger-rank-fork-family",
600
+ "status": "ran",
601
+ "deterministic": true
602
+ },
603
+ {
604
+ "id": "trigger-negative-4",
605
+ "kind": "trigger-negative",
606
+ "prompt": "Write a Vitest test for this plain utility function that formats a currency string",
607
+ "strictness": "high",
608
+ "trials": 1,
609
+ "passes": 1,
610
+ "passRate": 1,
611
+ "passAtK": 1,
612
+ "grader": "trigger-rank-fork-family",
613
+ "status": "ran",
614
+ "deterministic": true
615
+ },
616
+ {
617
+ "id": "trigger-negative-5",
618
+ "kind": "trigger-negative",
619
+ "prompt": "Fix this tsc error: Property 'total' does not exist on type 'OrderDraft'",
620
+ "strictness": "high",
621
+ "trials": 1,
622
+ "passes": 1,
623
+ "passRate": 1,
624
+ "passAtK": 1,
625
+ "grader": "trigger-rank-fork-family",
626
+ "status": "ran",
627
+ "deterministic": true
628
+ },
629
+ {
630
+ "id": "trigger-negative-6",
631
+ "kind": "trigger-negative",
632
+ "prompt": "Review this MobX store for accessibility-modifier and method-ordering issues",
633
+ "strictness": "high",
634
+ "trials": 1,
635
+ "passes": 1,
636
+ "passRate": 1,
637
+ "passAtK": 1,
638
+ "grader": "trigger-rank-fork-family",
639
+ "status": "ran",
640
+ "deterministic": true
641
+ },
642
+ {
643
+ "id": "assert-after-async-action",
644
+ "kind": "behavior",
645
+ "prompt": "This test is flaky:\n\n```ts\nit(\"loads items\", () => {\n store.fetchItems();\n expect(store.items).toHaveLength(3);\n});\n```\n\n`fetchItems` is an async action. Why is this flaky and how do I fix it?",
646
+ "strictness": "high",
647
+ "trials": 10,
648
+ "passes": 10,
649
+ "passRate": 1,
650
+ "passAtK": 1,
651
+ "grader": "judge",
652
+ "status": "ran",
653
+ "trialRecords": [
654
+ {
655
+ "output": "The flakiness comes from the assertion running before `fetchItems` has actually finished. `store.fetchItems()` returns a promise (or otherwise resolves later), but the test calls it and immediately asserts on `store.items` on the next line, synchronously. At that moment the action has only just been kicked off — the `await` inside it hasn't resumed yet, so `items` is still its initial value (typically `[]`). The assertion fails.\n\nWhether it *sometimes* passes is timing-dependent: if the mocked dependency (e.g. the API stub) resolves on a microtask that happens to drain before the assertion in a particular run, `items` can be populated by the time you check — under different scheduling or load it won't be. That nondeterminism is the flake. It isn't a bug in the store; it's the test asserting at the wrong point in time.\n\nFix it by waiting for the action to actually complete before asserting. The cleanest form, if `fetchItems` is a plain `async` method, is to await its own promise:\n\n```ts\nit(\"loads items\", async () => {\n await store.fetchItems();\n expect(store.items).toHaveLength(3);\n});\n```\n\nIf `fetchItems` doesn't return a promise you can await (e.g. it's a `runInAction`-wrapped action that internally awaits but doesn't return the chain), wait on the condition instead of the call:\n\n```ts\nit(\"loads items\", async () => {\n store.fetchItems();\n await when(() => store.items.length === 3);\n expect(store.items).toHaveLength(3);\n});\n```\n\nNotes that matter here:\n\n- Prefer awaiting the action's promise or a `when(...)` predicate. Do **not** reach for a fixed `setTimeout(resolve, 100)` or a bare `await Promise.resolve()` before asserting — a fixed sleep is slower than necessary and still nondeterministic under CI load, which just reintroduces the flake with extra latency.\n- Keep `enforceActions` at the project's real setting; don't loosen it to make setup easier. If you need to seed precondition state for the test, do it through `runInAction`, not by mutating observables directly.\n- After the fix, run the test a few times in a row and confirm it's deterministic — if it still occasionally fails, there's a second missing wait (e.g. a reaction that updates `items` after the action resolves), and that reaction's effect needs its own `when(...)` before the assertion.\n- Don't mark it skipped or `.only` to make the suite green; the correct fix is the missing `await`/`when`, and the correct reason to change the test is \"the assertion ran before the action's post-await state existed.\"",
656
+ "outputSha256": "84cef55aa07f506fb02fa20621266f9bb300eec3c3a3f2e9958a76d16c9f560d",
657
+ "promptSha256": "256c7a29d48da077583a34f8c3535e155a89832681b4cf637fdc4f6e074562e5",
658
+ "deterministic": [],
659
+ "judge": {
660
+ "verdict": "pass",
661
+ "reason": "Answer explains the assertion runs synchronously right after fetchItems, before the action's internal await resumes, so items is still initial/empty (timing-dependent flake). It shows corrected code: `await store.fetchItems();` before expect, plus a `when(() => store.items.length === 3)` alternative. It explicitly warns against fixed setTimeout/sleep rather than recommending it, so no fail criteria hold."
662
+ },
663
+ "passed": true
664
+ },
665
+ {
666
+ "output": "This isn't actually flaky in the \"sometimes passes, sometimes fails\" sense — it's a test that asserts on state **before the action it just called has had any chance to finish**. In practice it fails almost always (or only passes by accident if a previous test already populated `store.items`), and any appearance of flakiness is an illusion: the assertion is racing the async action, and whether it \"passes\" depends entirely on leftover state from elsewhere, not on this test's own action completing.\n\n**Why it fails**\n\n`store.fetchItems()` returns a promise (or kicks off one internally). Calling it starts the async work, but execution returns to the test synchronously at the first `await` inside the action — before the network/mock resolves and before any `runInAction` writes the fetched data into `store.items`. So `expect(store.items).toHaveLength(3)` runs against the store's initial (empty or stale) state. If another test earlier in the file happened to leave three items on a shared store instance, this one \"passes\" — which is exactly the kind of order-dependent result that looks like flakiness.\n\n**How to fix it**\n\nMake the test async and wait for the action's own completion before asserting:\n\n```ts\nit(\"loads items\", async () => {\n await store.fetchItems();\n expect(store.items).toHaveLength(3);\n});\n```\n\nThis works if `fetchItems` is a plain `async` method whose returned promise resolves after the state is written — which is the shape `mobx-store-implementation` / `rules/patterns.mdc` should be giving you (the state assignment happens inside `runInAction` before the action's promise settles).\n\nIf `fetchItems` doesn't return a promise you can await (e.g. it's fire-and-forget, or it only exposes a `loading` flag), wait on a predicate instead:\n\n```ts\nit(\"loads items\", async () => {\n store.fetchItems();\n await when(() => !store.loading); // or: () => store.items.length === 3\n expect(store.items).toHaveLength(3);\n});\n```\n\n**What not to do**\n\n- Don't add a `setTimeout(resolve, 100)` / sleep before asserting. It's slower than awaiting the real completion and still not deterministic under CI load — the exact thing producing the flake.\n- Don't disable `enforceActions` (or mutate `store.items` directly) to seed the \"3 items\" so the test passes — that hides the race rather than fixing it.\n- Don't skip the test. The missing `await`/`when` is the bug.\n\n**Verification**\n\nAfter the fix, run the test 2–3 times in a row and, ideally, in isolation as well as in the full-file order — it should pass every time. If it only passes in file order, you still have a hidden dependency on a shared store instance leaking state between tests; give each test its own store fixture if so.",
667
+ "outputSha256": "6b53489b1b094e7a03c1533dd72e499854f065f614c6209043c451d4a40bede3",
668
+ "promptSha256": "256c7a29d48da077583a34f8c3535e155a89832681b4cf637fdc4f6e074562e5",
669
+ "deterministic": [],
670
+ "judge": {
671
+ "verdict": "pass",
672
+ "reason": "Answer explains the assertion runs synchronously before the async action's await resolves and writes store.items, so it sees stale/empty state (\"execution returns to the test synchronously at the first await... before any runInAction writes\"). It shows corrected code with `await store.fetchItems();` before expect, plus a `when(() => !store.loading)` predicate alternative. It explicitly warns against setTimeout/sleep rather than recommending it."
673
+ },
674
+ "passed": true
675
+ },
676
+ {
677
+ "output": "# Why it's flaky\n\n`fetchItems` is async, so the line right after it runs **before the action has actually done its work**. An async action yields at its first `await` — typically the API call inside:\n\n```ts\nasync fetchItems() {\n const items = await this.api.getItems();\n runInAction(() => { this.items = items; });\n}\n```\n\nWhen you call `store.fetchItems()`, control returns to your test at that first `await`, with `this.items` still holding its initial value. So `expect(store.items).toHaveLength(3)` is asserting against state the action hasn't set yet.\n\nWhy \"flaky\" rather than \"always failing\": whether your synchronous assertion happens to see the settled state depends on microtask/macrotask timing and how the mocked API resolves. If the mock resolves synchronously (e.g. a stub returning an already-resolved promise) the state may already be set by the time the assertion runs — under a real network delay or a different mock it won't be. Same test, different timing, different result.\n\n# How to fix it\n\nMake the test `async` and **wait for the action's own completion** before asserting. `fetchItems` returns a promise, so await it directly:\n\n```ts\nit(\"loads items\", async () => {\n await store.fetchItems();\n expect(store.items).toHaveLength(3);\n});\n```\n\nThat's the preferred fix — you're waiting on the exact thing whose completion you care about, no guessing.\n\nIf the action does **not** return a promise (it fires the request but you have no handle to it), wait on a predicate instead:\n\n```ts\nit(\"loads items\", async () => {\n store.fetchItems();\n await when(() => store.items.length === 3);\n expect(store.items).toHaveLength(3);\n});\n```\n\n`when` resolves as soon as the predicate becomes true, so it's both correct and faster than a sleep.\n\n# What not to do\n\nDo **not** \"fix\" the flake with a fixed delay:\n\n```ts\nit(\"loads items\", async () => {\n store.fetchItems();\n await new Promise((r) => setTimeout(r, 100));\n expect(store.items).toHaveLength(3);\n});\n```\n\nThis is slower than necessary and still not deterministic — under CI load the 100ms may not be enough, which is exactly how this kind of flake comes back. It also masks whether the action even works correctly; waiting on the promise or a `when(...)` predicate proves the action actually completed.\n\n# Also worth checking\n\n- If `fetchItems` returns `void` and only mutates state, wrap it so it returns the promise (`return runInAction(...)` inside the async method, or return the awaited call), so callers — including tests — can await it.\n- Keep `enforceActions` at its real setting when seeding any fixture state; don't loosen it to set `store.items` directly for this test.\n- Run the fixed test 2–3 times in a row locally to confirm it's deterministic, not just passing once.",
678
+ "outputSha256": "225cb4fab58f5b8f152da4c254cb19fd92b12cb668be8e598e3a7afb33fc6343",
679
+ "promptSha256": "256c7a29d48da077583a34f8c3535e155a89832681b4cf637fdc4f6e074562e5",
680
+ "deterministic": [],
681
+ "judge": {
682
+ "verdict": "pass",
683
+ "reason": "Answer explains the assertion runs before the async action's first await resolves, leaving store.items at its initial value (\"control returns to your test at that first await, with this.items still holding its initial value\"). It shows the corrected test with `await store.fetchItems();` before the expect, and also a `when(() => store.items.length === 3)` predicate variant. It explicitly warns against a fixed setTimeout delay rather than recommending it."
684
+ },
685
+ "passed": true
686
+ },
687
+ {
688
+ "output": "This test is flaky because it asserts **synchronously** immediately after calling an async action, but the action's state mutation happens only after its internal `await` resolves.\n\n## Why it's flaky\n\n`fetchItems()` is async. When you call it, it runs synchronously up to its first `await` (e.g. `await api.getItems()`), then returns a promise and yields. At that moment `store.items` is still whatever it was before — empty, or stale from a previous test. Your `expect(...).toHaveLength(3)` therefore runs against pre-fetch state.\n\nWhether the test passes depends on *when* the promise's continuation lands relative to the assertion:\n\n- If the mocked API resolves synchronously (or the microtask queue drains before the assertion), it might pass.\n- If the mock has any real async delay, or the runner's scheduling differs, it fails.\n\nThat's the flakiness: it's a race between the action's continuation and the synchronous assertion, and it's decided by scheduling, not by correctness. It may even pass locally and fail in CI under load.\n\n## The fix\n\nWait for the async action to actually complete before asserting. Two correct options, in order of preference:\n\n**1. Await the action's returned promise** (cleanest, when `fetchItems` is a plain `async` method that resolves after it sets state):\n\n```ts\nit(\"loads items\", async () => {\n await store.fetchItems();\n expect(store.items).toHaveLength(3);\n});\n```\n\n**2. Await a predicate on the state itself** (use when the action doesn't resolve after the mutation, or fires a fire-and-forget internal call):\n\n```ts\nimport { when } from \"mobx\";\n\nit(\"loads items\", async () => {\n store.fetchItems();\n await when(() => store.items.length === 3);\n expect(store.items).toHaveLength(3);\n});\n```\n\nEither way, the test now waits on the action's own completion (its promise or a `when(...)` on the resulting state) rather than guessing at timing.\n\n## What not to do\n\n- **Don't** add `setTimeout(resolve, 100)` or `await new Promise(r => setTimeout(r, 100))`. It's slower than necessary and still not deterministic — under CI load 100ms may not be enough, and you're back to a flake.\n- **Don't** loosen `enforceActions` (e.g. to `'never'`) to \"fix\" it. The problem isn't an action-enforcement constraint; it's a missing `await`. If you need to seed precondition state, use `runInAction`, not a looser config.\n- **Don't** mark the test `.skip` or `.fails`. That hides a real timing bug in the test (and possibly reveals a genuine bug in the store if the action never resolves after mutating).\n\n## Also check\n\nIf this test tends to fail with stale data from a *previous* test (e.g. items left over at length 3 from an earlier case), the wait alone won't fix it — make sure each test constructs its own store instance with freshly mocked dependencies rather than sharing a store across tests. The fix above assumes a fresh store; combined with a proper wait, it becomes deterministic.",
689
+ "outputSha256": "695c8d2a6f028631dbb5050932d516d62e8d721596d4b72013dcf3de7a20350d",
690
+ "promptSha256": "256c7a29d48da077583a34f8c3535e155a89832681b4cf637fdc4f6e074562e5",
691
+ "deterministic": [],
692
+ "judge": {
693
+ "verdict": "pass",
694
+ "reason": "Answer explains the assertion runs synchronously before the action's internal await resolves, so store.items is stale/empty (\"asserts synchronously immediately after calling an async action... state mutation happens only after its internal await resolves\"). It shows corrected code: `await store.fetchItems();` before expect, and a `when(() => store.items.length === 3)` predicate variant. It warns against setTimeout rather than recommending it, so no fail criteria hold."
695
+ },
696
+ "passed": true
697
+ },
698
+ {
699
+ "output": "The test isn't actually flaky in the \"sometimes passes\" sense — it's wrong deterministically. It only ever appears to work when the async action happens to have already resolved before the assertion runs (e.g. a mocked API that resolves synchronously, or a stale `store.items` left over from a previous test). Under any real async dependency it fails; under a fast/synchronous mock it passes. That inconsistency across mocks, runners, and test ordering is what reads as \"flaky.\"\n\nThe root cause: `store.fetchItems()` returns before the pending promise inside it settles. Nothing at the await boundary has run yet, so `store.items` is still its pre-fetch value (usually `[]`). You're asserting on state that the action hasn't set yet.\n\nThe fix is to wait for the action's own completion before asserting. Two correct shapes:\n\n**If `fetchItems` is a plain `async` method, await it directly:**\n\n```ts\nit(\"loads items\", async () => {\n await store.fetchItems();\n expect(store.items).toHaveLength(3);\n});\n```\n\n**If it isn't async-returning (fire-and-forget, or returns void), wait on the predicate instead:**\n\n```ts\nimport { when } from \"mobx\";\n\nit(\"loads items\", async () => {\n store.fetchItems();\n await when(() => store.items.length > 0);\n expect(store.items).toHaveLength(3);\n});\n```\n\nA few things tied to this:\n\n- Don't reach for `setTimeout(resolve, 100)` or a fixed sleep. It's slower than necessary and still not deterministic — under CI load the 100ms can elapse before the microtask chain finishes. Wait on the promise or a `when(...)` predicate.\n- Don't loosen `enforceActions` in the test file to set up fixtures. If you need to seed precondition state, do it through `runInAction` so you're not hiding the same bug class the real app would hit.\n- Mark the test callback `async` and `await` the wait — a bare `when(...)` call without `await` reintroduces the exact bug you're fixing.\n\nIf, after adding the correct wait, the test still occasionally fails to reach the condition, that points at a real timing/ordering bug in the store's action itself (e.g. overlapping fetches clobbering `items`), not at the test — fix the source rather than re-skipping or re-loosening the test.",
700
+ "outputSha256": "a5dd4ac9a28cd71563b7b825d5d8eba7eceabcc95f2cc8a9e457560b182bc7c3",
701
+ "promptSha256": "256c7a29d48da077583a34f8c3535e155a89832681b4cf637fdc4f6e074562e5",
702
+ "deterministic": [],
703
+ "judge": {
704
+ "verdict": "pass",
705
+ "reason": "Answer states fetchItems returns before its promise settles so store.items is still pre-fetch/[] when asserted (criterion 1). It shows corrected code: `await store.fetchItems();` before expect, and a `when(() => store.items.length > 0)` predicate variant (criterion 2). It explicitly warns against setTimeout/sleep, not recommending it, so no fail criteria hold."
706
+ },
707
+ "passed": true
708
+ },
709
+ {
710
+ "output": "It's flaky because `fetchItems()` is async — it returns before its post-`await` code runs. The assertion executes synchronously, right after the call, while the action is still suspended at its first `await`. Depending on timing (how fast the mocked API resolves, microtask scheduling, machine load), `store.items` may be empty, partially populated, or already set. Sometimes the promise happens to resolve fast enough that the assertion passes; sometimes it doesn't.\n\nSo the test is racing the action, and the race resolves differently run to run.\n\nThe fix is to wait for the action's own completion before asserting, not to assert synchronously.\n\n**If `fetchItems` returns a promise** (plain `async` method):\n\n```ts\nit(\"loads items\", async () => {\n await store.fetchItems();\n expect(store.items).toHaveLength(3);\n});\n```\n\n**If it doesn't return a usable promise** (e.g. it sets a `loading` flag and you don't want to rely on its return value), wait on a predicate that represents \"the action reached the state you're asserting on\":\n\n```ts\nimport { when } from \"mobx\";\n\nit(\"loads items\", async () => {\n store.fetchItems();\n await when(() => store.items.length === 3);\n expect(store.items).toHaveLength(3);\n});\n```\n\nNote the `when` predicate needs to be something that becomes true once the action finishes; don't make it identical to the assertion in a way that would pass vacuously before the action runs (e.g. if `items` starts non-empty). Asserting on an explicit loaded/`loading` flag at the end can be clearer.\n\n**What not to do:** don't add a `setTimeout(resolve, 100)` or similar sleep before the assertion. It's both slower than necessary and still nondeterministic under CI load — you're just hoping the sleep outlasts the action. Waiting on the action's promise or a `when(...)` predicate is deterministic and faster.\n\nAlso worth checking: if `fetchItems` mutates observables after the `await` without an action wrapper, and the project runs with `enforceActions` enabled, you may hit a separate \"state mutation outside an action\" error once you start awaiting. In that case seed through `runInAction` in the store's action (the correct fix), not by loosening `enforceActions` in the test file — that would hide the same bug the real app hits.\n\nAfter fixing, run the test a few times in a row locally to confirm it's stable.",
711
+ "outputSha256": "b0b215ce2c37617472e19deca55d717e2363fb6a113639cd35e360aa63f0aa96",
712
+ "promptSha256": "256c7a29d48da077583a34f8c3535e155a89832681b4cf637fdc4f6e074562e5",
713
+ "deterministic": [],
714
+ "judge": {
715
+ "verdict": "pass",
716
+ "reason": "Answer explains the assertion runs synchronously while the action is suspended at its first await, so store.items may be stale/empty (\"The assertion executes synchronously, right after the call, while the action is still suspended at its first await\"). It shows corrected code with `await store.fetchItems();` before expect, and a `when(() => store.items.length === 3)` predicate alternative. It explicitly warns against setTimeout/sleep rather than recommending it."
717
+ },
718
+ "passed": true
719
+ },
720
+ {
721
+ "output": "# Why this is flaky\n\n`fetchItems()` is an async action: it kicks off, hits the first `await` (the API call), and returns a pending promise **immediately**. The next line runs before the action has resumed past its `await` and assigned `store.items`.\n\nSo the assertion executes against whatever `store.items` was *before* the fetch resolved — the initial `[]` (or stale data). It sometimes \"passes\" only because a prior test, a mock that resolves synchronously, or a microtask that happens to have drained in between left the store already populated. Change test order, CPU load, or mock timing and the assertion reads stale state and fails. The failure looks random but the root cause is fixed: **the test never waits for the action it just started.**\n\n# The fix\n\nWait for the action's own completion before asserting. Since `fetchItems` is a plain `async` method, await its returned promise:\n\n```ts\nit(\"loads items\", async () => {\n await store.fetchItems();\n expect(store.items).toHaveLength(3);\n});\n```\n\nIf `fetchItems` doesn't return a promise (e.g. it's a `flow`/generator action, or returns `void` and only updates state internally), wait on a predicate for the condition the action should reach, instead of the promise:\n\n```ts\nimport { when } from \"mobx\";\n\nit(\"loads items\", async () => {\n store.fetchItems();\n await when(() => store.items.length === 3);\n expect(store.items).toHaveLength(3);\n});\n```\n\nOr, if it sets a loading/status flag, that predicate is often the cleaner wait:\n\n```ts\nawait when(() => !store.isLoading);\n```\n\n# What not to do\n\n- **Don't** paper over it with a fixed sleep (`setTimeout(resolve, 100)`). It's slower than needed and still non-deterministic under CI load.\n- **Don't** loosen `enforceActions` or mutate `store.items` directly in the test to \"pre-seed\" the result. Seed fixture state through `runInAction`, not by disabling action enforcement.\n- **Don't** skip the test — that just hides the missing `await`.\n\n# Verify\n\nMake it `async`, await the action's completion (promise or `when(...)`), then run the suite and re-run a few times locally in a row. If it's still flaky, there's a second missing wait somewhere (often a reaction that fires the state change one microtask later) — chase that, don't add a timeout.",
722
+ "outputSha256": "d136ddb77d8625a56754052fcc0f6c116a195e8c7b517f7e4f9c7b42cc3e99a9",
723
+ "promptSha256": "256c7a29d48da077583a34f8c3535e155a89832681b4cf637fdc4f6e074562e5",
724
+ "deterministic": [],
725
+ "judge": {
726
+ "verdict": "pass",
727
+ "reason": "Answer explains fetchItems returns a pending promise immediately and the assertion runs before the await resumes and assigns store.items, reading stale/empty state. It shows corrected code: `await store.fetchItems();` before expect, plus a `when(() => store.items.length === 3)` alternative. It explicitly warns against fixed setTimeout/sleep rather than recommending it."
728
+ },
729
+ "passed": true
730
+ },
731
+ {
732
+ "output": "The test is flaky because it's asserting synchronously on state that `fetchItems` sets only after an `await` inside it. When you call `store.fetchItems()`, the action starts running, hits its first `await`, and returns control to your test immediately — the state update (`store.items = ...`, presumably inside a `runInAction` or a later `await` block) hasn't happened yet. The assertion then runs against the store's *initial* state.\n\nWhy it's *flaky* rather than consistently failing: if the mocked API stub resolves synchronously enough (e.g. the promise is already settled in the microtask queue in some setups), the continuation can occasionally land before your assertion executes; under different timing/load, it doesn't. That's the classic signature of asserting on post-`await` state without awaiting the action.\n\n**The fix: await the action itself.**\n\n```ts\nit(\"loads items\", async () => {\n await store.fetchItems();\n expect(store.items).toHaveLength(3);\n});\n```\n\n`fetchItems` is a plain `async` method, so it returns a promise that resolves only after the action's body (including its `await`ed fetch and the state assignment) has fully completed. Awaiting that promise guarantees the state is set before you assert.\n\n**If `fetchItems` doesn't return a promise you can await** — e.g. it fires and forgets, or returns early — then wait on the condition instead of the call:\n\n```ts\nit(\"loads items\", async () => {\n store.fetchItems();\n await when(() => store.items.length > 0);\n expect(store.items).toHaveLength(3);\n});\n```\n\n`when` from MobX resolves as soon as the predicate becomes true, so you're waiting on exactly the state change you care about.\n\n**What not to do:**\n\n- Don't add a fixed `setTimeout(resolve, 100)` or a sleep. It's slower than necessary and still not deterministic — under a loaded CI machine the fetch can take longer than 100ms and you'll flake again.\n- Don't loosen `enforceActions` in the test to set up fixture state directly. If you need seed data, do it through `runInAction`.\n\nOne more thing worth checking while you're in there: if `store.fetchItems` owns any `autorun`/`reaction`, add (or keep) a test that calls `dispose()` and then triggers the mutation that would normally fire the reaction, asserting the effect does *not* run again. That's a separate failure mode from this one, but it's the other thing that tends to go untested in stores like this.",
733
+ "outputSha256": "3f2d9242bced8ec5f9e1a21a73d4684ae6f26bab46d09f0961ef650d2c881c71",
734
+ "promptSha256": "256c7a29d48da077583a34f8c3535e155a89832681b4cf637fdc4f6e074562e5",
735
+ "deterministic": [],
736
+ "judge": {
737
+ "verdict": "pass",
738
+ "reason": "Answer explains the assertion runs synchronously before the action's internal await resolves and state update happens (\"hits its first await, and returns control... hasn't happened yet\"), and shows corrected code with `await store.fetchItems();` before expect, plus a `when(() => store.items.length > 0)` alternative. It explicitly warns against setTimeout/sleep rather than recommending it."
739
+ },
740
+ "passed": true
741
+ },
742
+ {
743
+ "output": "The test is flaky because `fetchItems()` is async, so `store.fetchItems()` returns (and the test line finishes) as soon as the function hits its first `await` — before the state update that sets `store.items` has run. `expect(store.items)` is then executed against whatever `items` happened to be at that exact moment: usually empty, but it could occasionally have leftover state from a previous test, or (rarely, under different scheduling) already be populated. That timing-dependent race is the flake.\n\n## Fix\n\nWait for the async action's own completion before asserting. If `fetchItems` is a plain `async` method, just await its returned promise:\n\n```ts\nit(\"loads items\", async () => {\n await store.fetchItems();\n expect(store.items).toHaveLength(3);\n});\n```\n\nIf `fetchItems` doesn't return a promise (e.g. it's a fire-and-forget action that kicks off work internally), wait on a predicate that the action is expected to satisfy, using MobX's `when`:\n\n```ts\nimport { when } from \"mobx\";\n\nit(\"loads items\", async () => {\n store.fetchItems();\n await when(() => store.items.length === 3);\n expect(store.items).toHaveLength(3);\n});\n```\n\nEither way, the assertion now runs only after the state it depends on has actually been set.\n\n## What not to do\n\n- Don't insert a fixed `setTimeout(resolve, 100)` (or similar sleep) before the assertion. It's slower than necessary and still not deterministic — under CI load the action can take longer than the sleep, and the test flakes again.\n- Don't loosen `enforceActions` in the test file to mutate `store.items` directly as a shortcut. Keep it at the project's real setting and seed any precondition state through `runInAction` instead.\n- Don't skip or delete the test to make the suite green. The missing `await`/`when` is the actual bug in the test; fix that.\n\nIf the store also owns a `reaction`/`autorun` whose effect this test relies on, apply the same rule there: trigger the mutation, then wait for the reaction's effect (via `when` or a microtask tick) before asserting it fired.",
744
+ "outputSha256": "4dbaa652cd78684b5333ef5c74574cf5fb42fc3315de2209f326fcb8837d74f3",
745
+ "promptSha256": "256c7a29d48da077583a34f8c3535e155a89832681b4cf637fdc4f6e074562e5",
746
+ "deterministic": [],
747
+ "judge": {
748
+ "verdict": "pass",
749
+ "reason": "Explains flake: fetchItems returns at first await before state update sets store.items, so expect runs against stale/empty state. Shows corrected code with `await store.fetchItems();` before expect, plus a `when(() => store.items.length === 3)` alternative. Warns against setTimeout sleep rather than recommending it."
750
+ },
751
+ "passed": true
752
+ },
753
+ {
754
+ "output": "The test is flaky because `store.fetchItems()` returns before the async action's `await` has resolved. The assertion on `store.items` runs synchronously, right after the call is kicked off — so it races the microtask/promise that actually sets `items`. Sometimes the stub resolves fast enough that the assertion happens to pass; under load or a slower mock it reads the pre-fetch state (empty or stale `items`) and fails. The `toHaveLength(3)` assertion isn't wrong, the timing is.\n\nFix: wait for the action's own completion before asserting.\n\nIf `fetchItems` is a plain `async` method that returns a promise, just await it:\n\n```ts\nit(\"loads items\", async () => {\n await store.fetchItems();\n expect(store.items).toHaveLength(3);\n});\n```\n\nIf it swallows its promise (doesn't return the awaitable), wait on the condition the action should reach instead:\n\n```ts\nit(\"loads items\", async () => {\n store.fetchItems();\n await when(() => store.items.length === 3);\n expect(store.items).toHaveLength(3);\n});\n```\n\nEither way, do not paper over it with a fixed `setTimeout(resolve, 100)` — that's slower than needed and still not deterministic under load, since it just moves the race rather than removing it.\n\nA couple of related points while you're in this test:\n\n- Keep `enforceActions` at the project's real setting. Making the test async to wait for the action is fine; loosening `enforceActions` in the test file to set up fixtures is not — seed precondition state through `runInAction`.\n- Once it's `async`, make sure you actually `await` the assertion path (the `it` callback should be `async` and the `await` should be in it), otherwise you've introduced the same fire-and-forget race one level up.\n- Re-run it 2–3 times locally after the fix to confirm the flake is gone, not just that it passed once.",
755
+ "outputSha256": "6242bb32758af97f818a525ea9b0c6d33616b62b38b009fd71db59be1c45fcc1",
756
+ "promptSha256": "256c7a29d48da077583a34f8c3535e155a89832681b4cf637fdc4f6e074562e5",
757
+ "deterministic": [],
758
+ "judge": {
759
+ "verdict": "pass",
760
+ "reason": "Answer explains the assertion runs synchronously right after fetchItems() before the async await resolves, racing the promise that sets items (stale/empty state). It shows corrected code: `await store.fetchItems();` before expect, and an alternative `await when(() => store.items.length === 3);`. It explicitly warns against fixed setTimeout, not recommending it. Both pass criteria met, no fail criteria."
761
+ },
762
+ "passed": true
763
+ }
764
+ ]
765
+ },
766
+ {
767
+ "id": "dispose-test-effect",
768
+ "kind": "behavior",
769
+ "prompt": "I have a store with an autorun disposed in dispose(). My current test is:\n\n```ts\nit(\"disposes cleanly\", () => {\n store.dispose();\n});\n```\n\nDoes this actually prove the autorun stops running? If not, what should the test do instead?",
770
+ "strictness": "high",
771
+ "trials": 10,
772
+ "passes": 10,
773
+ "passRate": 1,
774
+ "passAtK": 1,
775
+ "grader": "judge",
776
+ "status": "ran",
777
+ "trialRecords": [
778
+ {
779
+ "output": "No, that test proves nothing. It only proves `dispose()` doesn't throw — a store whose `dispose()` forgets to call the disposer function, or disposes the wrong reaction, or never wired the autorun up in the first place, passes this test identically. It's a smoke test at best, and it's the exact \"just checks it doesn't throw\" pattern this skill flags.\n\nWhat it needs to instead is actually exercise the reaction after disposal:\n\n```ts\nit(\"disposes cleanly\", () => {\n const effect = vi.fn();\n const store = new MyStore(/* mocked deps */);\n // observe the thing the autorun is supposed to react to,\n // whose side effect is visible via `effect`\n const stop = autorun(() => {\n effect(store.someObservable);\n });\n\n // sanity: it runs while alive\n expect(effect).toHaveBeenCalled();\n\n store.dispose(); // disposes the store's OWN autorun\n const callsAfterDispose = effect.mock.calls.length;\n\n // trigger the mutation that would normally re-fire the reaction\n runInAction(() => { store.someObservable = \"changed\"; });\n\n expect(effect.mock.calls.length).toBe(callsAfterDispose);\n});\n```\n\nThe key move is: **mutate the observable the reaction depends on after `dispose()`, and assert the reaction's effect did NOT run again.** Just calling `dispose()` and stopping is not a test.\n\nTwo practical notes:\n\n1. If the store's autorun isn't directly observable from the outside, don't fake it with a spy on `dispose` — expose something the autorun does produce (a field it writes, a callback it invokes) so the test can observe the reaction's real effect. A private autorun with no observable effect is untestable by construction and usually signals the store should surface its effect somewhere.\n\n2. Don't add a `setTimeout`/sleep to \"let disposal take effect.\" `dispose()` and the disposer it calls are synchronous; if you find yourself needing a wait, something else is wrong — the mutation likely needs to be wrapped in `runInAction` (see `rules/testing.mdc`), not followed by a sleep.\n\nAlso worth keeping `enforceActions` at its real setting here: if setting `store.someObservable` outside an action throws under `enforceActions: 'always'`, that's the store telling you the mutation path is wrong — wrap it in `runInAction` rather than loosening the config in the test.",
780
+ "outputSha256": "b12c56abb710eb4b669a2c6fa785e276a9306879c069b2320b1df2946698696b",
781
+ "promptSha256": "8d25c0115b8b740b471efc9b1eb67840c8f387844415970bfe8dbd0e9fbf5a9e",
782
+ "deterministic": [],
783
+ "judge": {
784
+ "verdict": "pass",
785
+ "reason": "Answer states the test 'only proves dispose() doesn't throw' and that a store forgetting to call the disposer passes identically (criterion 1). It shows a corrected test: store.dispose(), then runInAction mutating store.someObservable, then asserting effect.mock.calls.length unchanged (criterion 2). No fail criteria: it denies sufficiency and never relies on store.disposed."
786
+ },
787
+ "passed": true
788
+ },
789
+ {
790
+ "output": "No — that test proves nothing about the autorun. `store.dispose()` is a synchronous method; as long as it doesn't throw, the test passes. It never observes whether the reaction was actually stopped, so it would still pass if `dispose()` did nothing at all, or if the autorun had never been registered in the first place. The test only exercises \"calling this method doesn't crash.\"\n\nWhat you actually need is a test where the autorun's *effect* is observable, then prove that effect stops happening after `dispose()`. Concretely:\n\n```ts\nit(\"stops the autorun after dispose\", async () => {\n const effect = vi.fn();\n\n // If the autorun is internal to the store, expose a way to observe it,\n // or assert on whatever state its effect actually mutates. Otherwise\n // build an equivalent reaction and dispose it the same way the store does.\n\n // 1. Prove the reaction fires before dispose.\n runInAction(() => { store.someObservable = 1; });\n await when(() => effect.mock.calls.length === 1);\n\n // 2. Dispose.\n store.dispose();\n\n // 3. Trigger the mutation that would normally fire it.\n runInAction(() => { store.someObservable = 2; });\n\n // 4. Give any (now-disposed) reaction a chance to run, then assert it didn't.\n await Promise.resolve();\n\n expect(effect).toHaveBeenCalledTimes(1); // still 1, not 2\n});\n```\n\nKey points:\n\n- **Establish the baseline first.** Assert the effect *did* run before `dispose()`. Without that, step 4 passing could just mean the reaction never worked.\n- **Trigger the same dependency after dispose.** `dispose()` should stop reactions that depend on `someObservable`; mutate that exact field, not an unrelated one.\n- **Wait before asserting the negative.** After the post-dispose mutation, yield a microtask (`await Promise.resolve()`) so a still-live synchronous reaction would have had its chance to run. Don't use a `setTimeout` sleep — a microtask flush is enough for MobX reactions and is deterministic.\n- **Test the observable effect, not the method call.** If the autorun's only job is an internal side effect (e.g. writing to a cache), assert on that side effect via a spy or the mutated state it produces.\n\nIf your store's autorun isn't externally observable, that's a signal the store is hard to test — either expose the side effect it mutates (which is usually state you'd want to assert on anyway) or inject a callback/spy dependency you can observe. The goal is that the test fails if `dispose()` is removed or the reaction is wired up incorrectly, which the current one-liner cannot do.",
791
+ "outputSha256": "5cc7194705139d86f11811b4535b8017ead5daf28b6997595eb635368831fba8",
792
+ "promptSha256": "8d25c0115b8b740b471efc9b1eb67840c8f387844415970bfe8dbd0e9fbf5a9e",
793
+ "deterministic": [],
794
+ "judge": {
795
+ "verdict": "pass",
796
+ "reason": "Answer states the test only proves dispose() doesn't throw and would pass even if dispose did nothing ('proves nothing about the autorun'). It shows a corrected test: baseline mutation, store.dispose(), post-dispose mutation of store.someObservable, then expect(effect).toHaveBeenCalledTimes(1) asserting the spy did not re-run. No fail criteria met."
797
+ },
798
+ "passed": true
799
+ },
800
+ {
801
+ "output": "No, that test doesn't prove anything about the autorun. `store.dispose()` returning without throwing only proves the call itself didn't blow up — it says nothing about whether the autorun was ever wired up correctly, whether it was actually disposed, or whether it still fires on subsequent mutations. A store that never registered the autorun, or that registered it but forgot to call the disposer, would pass this test identically.\n\nWhat the test should do instead: capture evidence that the autorun runs *before* dispose, then assert it does *not* run *after* dispose when you trigger the mutation that normally fires it.\n\n```ts\nit(\"stops the autorun after dispose()\", () => {\n const spy = vi.fn(); // or jest.fn()\n\n // Re-create the store so we control the autorun's side effect,\n // or expose a hook/observable the real autorun writes to.\n const store = new Store(apiMock);\n\n // 1. Prove the autorun actually fires while alive.\n runInAction(() => {\n store.relevantField = \"changed\"; // whatever the autorun watches\n });\n expect(store.autorunEffectCount).toBe(1); // or spy called / observable updated\n\n // 2. Dispose, then trigger the same mutation again.\n store.dispose();\n\n runInAction(() => {\n store.relevantField = \"changed-again\";\n });\n\n // 3. Assert the effect did NOT run a second time.\n expect(store.autorunEffectCount).toBe(1); // unchanged\n});\n```\n\nKey points on how to write this correctly:\n\n- **Assert the \"before\" case too.** Without step 1, a test that only checks the effect didn't fire after dispose would pass even if the autorun *never* worked — the test would be vacuously green. You need to show the reaction is live, then show dispose kills it.\n- **Trigger the exact mutation the autorun watches.** Disposing an autorun stops it from *reacting*; it doesn't stop the underlying observables from changing. So you must mutate the dependency and then check the autorun's *effect* didn't run — checking the observable changed proves nothing.\n- **Expose an observable effect to assert on.** The cleanest way is for the store's autorun to write to an observable or a counter you can read (or accept an injected callback/spy). If the autorun currently does something invisible (e.g. `console.log`), refactor it to write to an observable or take a subscriber you can spy on, so it's testable.\n- **Wait if the reaction is async.** If the autorun triggers async work, `await when(() => store.effectRan)` (or await the promise it kicks off) before asserting it ran — and after dispose, await a microtask tick (`await Promise.resolve()`) before asserting it did *not* run, so you're not asserting on a race. Don't replace this with a fixed `setTimeout`.\n- **Don't assert synchronously right after the mutation** if the reaction runs on a later tick — you'll get a false \"it didn't fire\" and a flaky test.\n\nAlso worth noting: if `dispose()` is supposed to tear down *multiple* reactions (autorun + reaction + when), add a case per reaction, or a single test that triggers each one's dependency and asserts each effect stayed frozen. And a \"disposes cleanly\" test is still fine to keep as a smoke test alongside this one — it just isn't sufficient on its own.",
802
+ "outputSha256": "ae959f24f739fd15da93145028326be3d46654f5e27ffe8952cd51327dec20a3",
803
+ "promptSha256": "8d25c0115b8b740b471efc9b1eb67840c8f387844415970bfe8dbd0e9fbf5a9e",
804
+ "deterministic": [],
805
+ "judge": {
806
+ "verdict": "pass",
807
+ "reason": "Answer states dispose() returning without throwing only proves the call didn't blow up, not that the autorun stops firing. It shows a corrected test: mutate store.relevantField, assert autorunEffectCount is 1, call store.dispose(), mutate again, assert count still 1 (effect did not re-run). No fail criteria hold."
808
+ },
809
+ "passed": true
810
+ },
811
+ {
812
+ "output": "No — that test proves nothing about the autorun. It only proves `dispose()` doesn't throw. A store can call `dispose()`, and even call `.dispose()` on some tracked reaction object, while still having a wiring bug: the autorun's disposer was never captured, was captured from the wrong object, or a second reaction was created and never disposed. All of those pass your current test.\n\nWhat you actually need to assert is behavioral: **after `dispose()`, the mutation that used to trigger the autorun no longer produces its effect.**\n\nA concrete shape:\n\n```ts\nit(\"stops the autorun after dispose()\", () => {\n const spy = vi.fn(); // or your runner's mock\n const store = new Store(deps, { onSomething: spy });\n\n // 1. Prove the autorun is live before dispose.\n runInAction(() => { store.something = \"x\"; });\n expect(spy).toHaveBeenCalledTimes(1);\n\n // 2. Dispose.\n store.dispose();\n\n // 3. Trigger the same mutation that fired it before.\n runInAction(() => { store.something = \"y\"; });\n\n // 4. The effect must NOT run again.\n expect(spy).toHaveBeenCalledTimes(1);\n});\n```\n\nKey points about this shape:\n\n- **A before-dispose trigger is mandatory.** Without step 1, step 4 is vacuous — a test that never saw the autorun fire can't tell \"disposed correctly\" from \"never wired up in the first place.\" The pre-dispose assertion is what makes the post-dispose assertion meaningful.\n- **Trigger the exact same mutation** the autorun observes, with a distinct value, so a passing test can't be explained by \"nothing changed.\"\n- **If the autorun's effect is itself asynchronous** (it awaits, or does I/O), you need a wait before each assertion — `await when(() => spy.mock.calls.length === 1)` for the pre-dispose check, and then a wait for a *quiet period* after the post-dispose mutation (e.g. `await when(() => false, { timeout: 50 })` or a `flushPromises()` tick) before asserting the count stayed at 1. Otherwise a slow async effect could be \"not yet run\" rather than \"stopped,\" and you'd get a false pass. Don't fix this with a bare `setTimeout` sleep in the passing case; only use a bounded wait in the *negative* direction, where you're asserting nothing happens.\n- **Don't loosen `enforceActions`** to seed the pre-dispose state directly; use `runInAction` as above.\n- **If the store owns multiple reactions**, either give each one its own observable it uniquely observes (so one test per reaction can isolate it), or assert the full set of effects and their call counts after dispose.\n\nSo: replace the throw-check with a fire-before / dispose / fire-after / assert-no-refire test. That's the one that actually catches a leaked or mis-wired disposer.",
813
+ "outputSha256": "25d30b1c7365fe19561041ceadc9bf48e61857a99cb60154ce32d1ad8264997a",
814
+ "promptSha256": "8d25c0115b8b740b471efc9b1eb67840c8f387844415970bfe8dbd0e9fbf5a9e",
815
+ "deterministic": [],
816
+ "judge": {
817
+ "verdict": "pass",
818
+ "reason": "Answer states the test only proves dispose() doesn't throw ('It only proves dispose() doesn't throw'), and gives a corrected test: spy effect, pre-dispose trigger, store.dispose(), post-dispose mutation of store.something, then expect(spy).toHaveBeenCalledTimes(1) asserting no re-fire. No fail criteria met."
819
+ },
820
+ "passed": true
821
+ },
822
+ {
823
+ "output": "No — that test proves nothing about the autorun. It only proves `dispose()` doesn't throw. A store can call `this.dispose = () => {}` (or have a `dispose` that never captured the reactions) and this test still passes. The whole point of `dispose()` is that the reaction *stops firing* afterward, and that's exactly the thing your assertion never touches.\n\n## What's missing\n\nYou need to observe the autorun's side effect before and after disposal, then trigger the mutation that would normally fire it and assert the effect did **not** happen again.\n\n```ts\nit(\"dispose() stops the autorun\", () => {\n const spy = vi.fn();\n // The autorun under test should call spy() when it fires.\n // If your store's autorun doesn't expose a hook, that's a signal\n // the store should accept an injected effect/callback for testability,\n // or you assert on the observable it writes instead.\n\n // 1. Prove it fires before dispose.\n store.someObservable = 1; // via runInAction if enforceActions\n expect(spy).toHaveBeenCalledTimes(1);\n\n // 2. Dispose.\n store.dispose();\n\n // 3. Trigger the mutation that would fire it.\n store.someObservable = 2;\n await Promise.resolve(); // let any microtask reaction settle\n\n // 4. Assert it did NOT fire again.\n expect(spy).toHaveBeenCalledTimes(1); // unchanged\n});\n```\n\nIf the autorun writes to an observable instead of calling a spy, assert on that observable in the same shape: it should reflect the pre-dispose mutation, not the post-dispose one.\n\n## A few things to get right\n\n- **Assert the effect, not the call.** `expect(store.dispose).not.toThrow()` is the same non-proof in different clothes. The test must hinge on observable behavior changing (or not changing) after disposal.\n- **Don't assert synchronously after the mutation if the reaction runs on a microtask.** Wait for the reaction to actually settle (`await Promise.resolve()` for a plain synchronous autorun, or `await when(() => <predicate>)`) before the post-dispose assertion — otherwise the \"didn't fire\" assertion can pass for the wrong reason, because the reaction hadn't had a chance to run yet in either case.\n- **Keep `enforceActions` at your project's real setting.** If you need to seed `someObservable` outside an action, wrap it in `runInAction`, don't flip `enforceActions` to `'never'`.\n- **Beware the false negative.** A test that is *supposed* to fail if dispose is broken is the only useful one. Consider temporarily commenting out the `dispose()` line to confirm the test actually goes red — if it stays green, your post-dispose assertion isn't observing the reaction at all.\n\n## Verify before calling it done\n\n- The test command passes.\n- The post-dispose assertion waits for the reaction to settle (no bare `setTimeout`).\n- Removing the `dispose()` call makes the test fail — that's the real proof it's testing disposal.\n- Running it 2–3 times in a row gives the same result.\n\nA store that owns `autorun`/`reaction`/`when` and exposes `dispose()` should always carry a test of this shape. The existing one-line version is the classic example of a passing test that would keep passing even if disposal were completely broken.",
824
+ "outputSha256": "7789b2ba5e7fb49fd5eae6e09fd51c12facb7a96ff5bfb92d800ef80dbc424c5",
825
+ "promptSha256": "8d25c0115b8b740b471efc9b1eb67840c8f387844415970bfe8dbd0e9fbf5a9e",
826
+ "deterministic": [],
827
+ "judge": {
828
+ "verdict": "pass",
829
+ "reason": "Answer states the test only proves dispose() doesn't throw, not that the autorun stops firing ('It only proves dispose() doesn't throw'). It shows a corrected test: spy before dispose, store.dispose(), then store.someObservable = 2, then expect(spy).toHaveBeenCalledTimes(1) unchanged. Also warns against asserting store.disposed or not.toThrow."
830
+ },
831
+ "passed": true
832
+ },
833
+ {
834
+ "output": "No, that test proves nothing about the autorun. It only proves that calling `store.dispose()` does not throw. A store can call `this.dispose = ...`-style cleanup that never actually calls the `autorun`'s returned disposer, or the autorun could be wired up to the wrong source, or `dispose()` could be a no-op entirely — and this test passes in every one of those cases. It has no assertion and never observes the reaction, so it cannot distinguish a correct dispose from a broken one.\n\nWhat the test needs to do is: capture the autorun's effect (a spy/counter), trigger the mutation that normally fires it, confirm it fired while the store is alive, then `dispose()`, trigger that same mutation again, and assert the effect did NOT fire a second time.\n\nConcretely:\n\n```ts\nit(\"stops the autorun after dispose()\", () => {\n const spy = jest.fn();\n // Make the store's autorun write through a hookable seam so the test can\n // observe it — e.g. the autorun calls an injected onChanged callback, or\n // writes to an observable the test can read.\n const store = new Store({ onChanged: spy, api: mockApi });\n\n // Baseline: the reaction is live.\n runInAction(() => {\n store.someObservableThatAutorunTracks = 1;\n });\n expect(spy).toHaveBeenCalledTimes(1);\n\n store.dispose();\n\n // Same mutation that fired it before must now be a no-op for the reaction.\n runInAction(() => {\n store.someObservableThatAutorunTracks = 2;\n });\n expect(spy).toHaveBeenCalledTimes(1); // unchanged — autorun did not run again\n});\n```\n\nKey points, per `rules/testing.mdc`:\n\n- **Observe the effect, don't just call dispose.** The autorun must be given an observable side effect (a spy, a callback, a counter) that the test can assert on. If the autorun currently has no externally visible effect, that's a testability gap in the store — route it through an injected callback or an observable the test reads.\n- **Prove it fired before dispose.** Triggering the mutation while alive and asserting the effect ran rules out a false pass where the autorun never worked at all. Without that baseline, a broken autorun plus a broken dispose would still leave your \"did not fire after dispose\" assertion green.\n- **Trigger the same mutation again after dispose.** That is the actual assertion — the effect count must not increase.\n- **Wait for the reaction if it's not synchronous.** If the autorun's effect lands on a microtask, `await when(() => spy.mock.calls.length === 1)` (or `await Promise.resolve()`) before the baseline assertion, and after dispose wait a tick before asserting the count is still 1. Don't use a fixed `setTimeout`/sleep.\n- **Keep `enforceActions` at the project setting.** Seed mutations through `runInAction`, as above — don't loosen it in the test file to set up the fixture.\n\nIf this test then fails — the spy fires again after `dispose()` — the bug is almost certainly that the disposer returned by `autorun(...)` isn't being called (or isn't stored/called) in `dispose()`. Fix the store, not the test.",
835
+ "outputSha256": "c72395b7ae6cc8e5b4edd8bed24a870a07ea526432277f5206a436ae1c3e5417",
836
+ "promptSha256": "8d25c0115b8b740b471efc9b1eb67840c8f387844415970bfe8dbd0e9fbf5a9e",
837
+ "deterministic": [],
838
+ "judge": {
839
+ "verdict": "pass",
840
+ "reason": "Answer states the test only proves dispose() doesn't throw and never observes the reaction (\"It only proves that calling store.dispose() does not throw... never observes the reaction\"). It shows a corrected test: baseline mutation asserts spy called once, then store.dispose(), then same mutation again, asserting spy count unchanged (\"autorun did not run again\"). No fail criteria hold."
841
+ },
842
+ "passed": true
843
+ },
844
+ {
845
+ "output": "No — that test proves nothing about the autorun. It only proves that `dispose()` runs without throwing. A store's `dispose()` can be a complete no-op, or can call `this.disposeAutorun()` on the wrong reference, or can dispose *after* the autorun was already rebuilt elsewhere, and this test would still pass. The actual leak — the autorun continuing to fire after the store is \"disposed\" — goes undetected.\n\nThe test needs to assert on the autorun's *effect*: it must have run before `dispose()`, and must not run again after it.\n\n## What the test should do instead\n\nSketch (adjust to the store's real API and the reaction's real side effect):\n\n```ts\nit(\"stops the autorun after dispose()\", () => {\n const spy = vi.fn();\n\n // 1. Prove the autorun is live: mutate the observable it tracks\n // and wait for its effect to actually run.\n store.setSomething(1);\n await when(() => spy.mock.calls.length > 0);\n // (the store's autorun under test must be the one calling `spy`,\n // e.g. it invokes a spy-able injected dependency)\n\n // 2. Baseline, so a later increment is unambiguous.\n const callsBeforeDispose = spy.mock.calls.length;\n\n store.dispose();\n\n // 3. Trigger the very mutation that fired it the first time.\n store.setSomething(2);\n\n // 4. Give the reaction a fair chance to run if it were still live —\n // a microtask/tick, not a fixed sleep. A still-live autorun would\n // schedule and flush here.\n await Promise.resolve();\n\n // 5. The assertion that actually proves disposal.\n expect(spy.mock.calls.length).toBe(callsBeforeDispose);\n});\n```\n\n## Key points about the shape\n\n- **Establish liveness first.** Trigger the tracked observable before `dispose()` and wait (`await when(...)`) until the effect has provably run. Without this baseline, step 5 could pass simply because the autorun was never wired up correctly in the first place — which is a different bug.\n- **Dispose, then re-trigger the same mutation.** Re-firing the identical dependency change is what distinguishes \"stopped\" from \"happened to not fire this time.\"\n- **Wait after the re-trigger, but not with a sleep.** A still-live autorun schedules its effect on the MobX reaction queue; a microtask flush (`await Promise.resolve()`, or an explicit `await when(() => false)`-style drain if the store uses async reactions) gives it the chance to run. Do **not** use `setTimeout(resolve, 100)` — it is slower than necessary and still not deterministic under CI load. If the effect is genuinely asynchronous inside, wait on a `when(...)` predicate over the spy instead.\n- **If the autorun's side effect isn't directly observable**, inject a spy-able dependency (e.g. a mock API/logger the autorun calls) at store construction so the effect is assertable. Constructing the store with mocked dependencies is the right setup either way, per `rules/testing.mdc`.\n\n## What this still doesn't cover\n\nIf the store owns more than one disposable reaction, this test only proves the one you spied on stops. If you have separate `autorun`/`reaction`/`when` instances, assert each — or at minimum the one whose leak would matter. And keep `enforceActions` at the project's real setting while seeding any preconditions via `runInAction`; don't loosen it to make the fixture easier.\n\nSo: replace the current test with the before/after-effect assertion above, run it 2–3 times in a row locally to confirm it's deterministic, and only then call disposal actually verified.",
846
+ "outputSha256": "1d72ef81b0abf13ddf9f0ab171e15b37377111d90c07a29ccf12fa8c6a922c4a",
847
+ "promptSha256": "8d25c0115b8b740b471efc9b1eb67840c8f387844415970bfe8dbd0e9fbf5a9e",
848
+ "deterministic": [],
849
+ "judge": {
850
+ "verdict": "pass",
851
+ "reason": "Answer states the test only proves dispose() runs without throwing, not that the autorun stops firing ('proves nothing about the autorun... only proves that dispose() runs without throwing'). It shows a corrected test: baseline spy calls, store.dispose(), then store.setSomething(2) re-triggering the tracked mutation, then asserts spy.mock.calls.length unchanged. No fail criteria hold."
852
+ },
853
+ "passed": true
854
+ },
855
+ {
856
+ "output": "No, that test proves almost nothing. It only asserts that `dispose()` doesn't throw synchronously. It never triggers the mutation the autorun is supposed to react to, so it can't tell the difference between a store whose autorun is correctly torn down and one where `dispose()` does nothing to the reaction at all — a store where the autorun leaks forever would still pass this test.\n\nWhat it needs to do, per this skill's disposal step:\n\n1. Set up the observable state the autorun depends on.\n2. Trigger the mutation that would normally fire the autorun, and wait for the effect to actually run — confirming the autorun was live to begin with (a control assertion; without this, a test that \"passes\" after dispose could just be passing because the autorun never worked).\n3. Call `store.dispose()`.\n4. Trigger that same mutation again.\n5. Assert the autorun's effect did **not** run again (e.g. its spy/counter/output is unchanged), after waiting for the reaction to actually get a chance to run — otherwise you can't distinguish \"didn't fire\" from \"hadn't fired yet.\"\n\nSketch:\n\n```ts\nit(\"stops its autorun after dispose\", async () => {\n const effects: string[] = [];\n const store = createStore({\n onStoreChange: (v: string) => effects.push(v),\n });\n\n // Prove the autorun is live before dispose.\n store.setValue(\"a\");\n await when(() => effects.length === 1);\n expect(effects).toEqual([\"a\"]);\n\n store.dispose();\n\n // Trigger the same mutation again.\n store.setValue(\"b\");\n // Give the reaction a chance to run if it's still wired up.\n await when(() => true); // or await Promise.resolve() for a sync microtask reaction\n expect(effects).toEqual([\"a\"]); // no [\"b\"] — the autorun is dead\n});\n```\n\nA couple of caveats specific to this shape:\n\n- The \"wait\" in step 5 matters. If the autorun fires synchronously, asserting right after the mutation is fine; if it's deferred (e.g. an async effect inside the autorun), you need to wait for the condition the effect *would* set, with a timeout — not a bare `setTimeout`, which is nondeterministic and slow. The usual trick is `await when(() => effects.length > 1, { timeout: 50 })` wrapped in a try/catch expecting the timeout, since \"it never gets there\" is the pass condition.\n- Don't loosen `enforceActions` in the test file to make fixture setup easier — seed any precondition state through `runInAction` instead, so the test exercises the store under its real configuration.\n\nTwo honest gaps in what I told you above, since I don't have your actual store code: if the autorun reads through a computed or a `when` that's already been satisfied, re-triggering the same mutation may not be a valid second probe — you'd need to mutate whatever the autorun's *actual* tracked dependency is. And if `dispose()` is supposed to dispose a whole set of reactions (multiple autoruns, a `reaction`, a `when`), this test only covers the one you pointed at; each disposable reaction deserves its own \"still live before, silent after\" check, not one blanket \"doesn't throw.\"",
857
+ "outputSha256": "e185778cc19d7e9a74b84daa584c620507afacce426a67fc2158b982fb29cab2",
858
+ "promptSha256": "8d25c0115b8b740b471efc9b1eb67840c8f387844415970bfe8dbd0e9fbf5a9e",
859
+ "deterministic": [],
860
+ "judge": {
861
+ "verdict": "pass",
862
+ "reason": "Answer states the test only proves dispose() doesn't throw and can't distinguish a torn-down autorun from a leaked one. It shows a corrected test: control mutation before dispose, then store.dispose(), then store.setValue(\"b\") again, then expect(effects).toEqual([\"a\"]) asserting the tracked effect did not re-run. No fail criteria hold."
863
+ },
864
+ "passed": true
865
+ },
866
+ {
867
+ "output": "No, that test does not prove the autorun stops running. `dispose()` is presumed to call `autorun`'s disposer; calling it and asserting nothing only proves the method doesn't throw. A store can call `dispose()` without ever having wired the disposer correctly (or without the autorun's effect being something you can observe at all), and this test will still pass green while the reaction keeps firing forever in production — a classic leak that this test is blind to.\n\nThe test needs to observe the reaction's *effect*, not the dispose call itself. Concretely:\n\n1. **Capture the effect.** Make the autorun's side effect something assertable — e.g. it pushes to an array, increments a counter, or calls a mocked dependency. If the autorun's only effect is console-logging or touching something you can't inspect, refactor the store slightly so the effect is observable (or spy on whatever function it calls).\n\n2. **Prove it fires before dispose.** Trigger the mutation the autorun depends on, wait for the reaction to run, and assert the effect happened. Without this baseline, a test that asserts \"nothing happened after dispose\" would also pass on a reaction that was never firing in the first place — a false green.\n\n3. **Dispose, then re-trigger the same mutation**, and assert the effect did *not* run again (compare counts/array length before and after).\n\n4. **Wait correctly.** Autoruns run as a microtask after the triggering mutation by default (unless the author used `runInAction` batching or `{ scheduler }`). Don't assert synchronously right after mutating — `await Promise.resolve()` (or `await when(() => effectHappened)` for the positive case) so you're not racing the reaction. Do not use a fixed `setTimeout` sleep; it's slower and still not deterministic under load.\n\nSketch:\n\n```ts\nit(\"disposes cleanly\", async () => {\n const effects: string[] = [];\n // store's autorun is expected to push into `effects` when `store.query` changes\n const store = createStore({ onReaction: (q) => effects.push(q) });\n\n store.setQuery(\"first\");\n await when(() => effects.length === 1); // baseline: it does fire\n expect(effects).toEqual([\"first\"]);\n\n store.dispose();\n\n store.setQuery(\"second\");\n await Promise.resolve(); // let any (incorrectly) live reaction run\n await Promise.resolve();\n expect(effects).toEqual([\"first\"]); // did NOT fire again\n});\n```\n\nTwo pitfalls to avoid here specifically:\n\n- **Don't skip the baseline assertion.** If you only test \"nothing runs after dispose,\" a store whose autorun was broken/non-firing all along passes, and you've proven nothing.\n- **Don't loosen `enforceActions`** to seed the store's initial state directly; go through an action or `runInAction`, so this test exercises the same discipline the real app uses.\n\nGiven your current test, the fix is to convert it from \"doesn't throw\" to \"the reaction fires, then stops firing after dispose\" — and to make the autorun's effect observable if it currently isn't.",
868
+ "outputSha256": "64eeee506ac323411840699202008a00bd2ba23b8c4ef53ea5138e8dba306c1a",
869
+ "promptSha256": "8d25c0115b8b740b471efc9b1eb67840c8f387844415970bfe8dbd0e9fbf5a9e",
870
+ "deterministic": [],
871
+ "judge": {
872
+ "verdict": "pass",
873
+ "reason": "Answer states the test only proves dispose() doesn't throw, not that the autorun stops firing ('calling it and asserting nothing only proves the method doesn't throw'). It shows a corrected test: baseline mutation, dispose(), then setQuery('second') and asserts effects still equals ['first'] (effect did not re-run). No fail criteria met."
874
+ },
875
+ "passed": true
876
+ },
877
+ {
878
+ "output": "No, that test proves nothing about the autorun. It only demonstrates that `dispose()` doesn't throw. A store can pass that assertion while its autorun is still fully wired up and firing — for example if `dispose()` was never actually called on the autorun's disposer, if the wrong disposer was stored, or if the reaction was re-created after disposal. All of those would still leave `store.dispose()` returning normally.\n\n## What the test should do instead\n\nYou need to assert on the *effect* of the autorun. That means three phases: confirm it fires before disposal, dispose, then trigger the same mutation again and confirm it does **not** fire.\n\n```ts\nit(\"stops its autorun after dispose()\", async () => {\n const spy = vi.fn();\n // point the autorun at something observable so its effect is observable\n const store = new MyStore(apiMock);\n const stop = autorun(() => spy(store.someObservable));\n\n // 1. prove the reaction is live before disposal\n await when(() => spy.mock.calls.length > 0); // or a plain microtask await\n const callsBeforeDispose = spy.mock.calls.length;\n\n // 2. dispose\n store.dispose();\n\n // 3. trigger the mutation that used to fire it, and give reactions a chance to run\n runInAction(() => {\n store.someObservable = \"changed\";\n });\n await Promise.resolve(); // flush any queued microtask\n\n // 4. assert the effect did NOT run again\n expect(spy.mock.calls.length).toBe(callsBeforeDispose);\n});\n```\n\nThe key shape is: **fire it, dispose, fire it again, assert silence.** Without the \"fire it before dispose\" step you can't distinguish \"dispose worked\" from \"the reaction never worked in the first place\" — a green test that would also pass on a completely broken store.\n\n## Notes specific to testing disposal\n\n- **Assert on the reaction's observable effect, not on internal flags.** Don't reach into the store and check something like `store._disposed === true`; that just re-tests the assignment. Drive the real observable the autorun reads and watch whether the effect runs.\n- **If the autorun is internal** (created inside the store's constructor rather than by the test), you can't add a spy inside it directly. In that case expose the observable effect it produces — a counter it increments, a field it writes, a callback it invokes — and assert on that. If the autorun has no externally observable effect at all, that's itself a design smell worth flagging: it's untestable by construction.\n- **Wait for the reaction to run before asserting, in both directions.** After the pre-dispose mutation, wait for the effect (`await when(...)` or a microtask flush) before recording `callsBeforeDispose`. After the post-dispose mutation, flush again before asserting the count is unchanged — otherwise you might assert before a (buggy) still-live reaction has had a chance to fire, and pass for the wrong reason. Do not use a fixed `setTimeout` for either wait; it's slower and still not deterministic under load.\n- **Don't loosen `enforceActions`** to seed the post-dispose mutation; wrap it in `runInAction` as above so the test exercises the same discipline the app uses.\n\nAs a bonus, this version also catches the opposite failure: if the autorun never fired even before disposal, step 1's `await when(...)` will time out rather than letting the test pass vacuously.",
879
+ "outputSha256": "a9e3c4ca890dd799db0bc3a06df70c41eb02bf8b6f13d11eb657e41da24abb01",
880
+ "promptSha256": "8d25c0115b8b740b471efc9b1eb67840c8f387844415970bfe8dbd0e9fbf5a9e",
881
+ "deterministic": [],
882
+ "judge": {
883
+ "verdict": "pass",
884
+ "reason": "Answer states the test only proves dispose() doesn't throw and the autorun may still fire ('It only demonstrates that dispose() doesn't throw'). It shows a corrected test: spy autorun, dispose(), mutate store.someObservable in runInAction, then assert spy.mock.calls.length unchanged. Also warns against asserting internal flags like store._disposed."
885
+ },
886
+ "passed": true
887
+ }
888
+ ]
889
+ }
890
+ ],
891
+ "verdict": "pass",
892
+ "scope": "bundled",
893
+ "skillDigest": "7b62db4e1eb66f87c671573c8b41ae45ee56a7ba2670ad98ca9c7c138076e09d",
894
+ "catalogDigest": "fd9b6a086f61a996f761a68f9e7a58cde1ce62e121f276d70bbef8e05e372f4b",
895
+ "judgePromptVersion": "2026-09-25.1",
896
+ "runner": "deepseek",
897
+ "model": "deepseek-chat",
898
+ "runnerPromptVersion": "2026-09-25.1",
899
+ "recordedAt": "2026-09-25T15:59:40.539Z",
900
+ "judge": "deepseek",
901
+ "judgeModel": "deepseek-chat"
902
+ }
903
+ ]
904
+ }