@mrciphersmith/keryx 0.2.163 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. package/dist/cli.js +87904 -56917
  2. package/dist/core.js +28418 -18708
  3. package/package.json +2 -2
  4. package/src/gdgraph/affected-report.ts +141 -0
  5. package/src/gdgraph/build.ts +170 -23
  6. package/src/gdgraph/service.ts +6 -0
  7. package/src/gdgraph/staleness.ts +253 -45
  8. package/src/gdskills/bundled/agents/codebase-navigator.md +55 -0
  9. package/src/gdskills/bundled/agents/design-advisor.md +64 -0
  10. package/src/gdskills/bundled/agents/docs-maintainer.md +56 -0
  11. package/src/gdskills/bundled/agents/end-to-end-tester.md +56 -0
  12. package/src/gdskills/bundled/agents/error-path-auditor.md +57 -0
  13. package/src/gdskills/bundled/agents/go-build-fixer.md +52 -0
  14. package/src/gdskills/bundled/agents/go-code-auditor.md +49 -0
  15. package/src/gdskills/bundled/agents/performance-auditor.md +63 -0
  16. package/src/gdskills/bundled/agents/python-build-fixer.md +52 -0
  17. package/src/gdskills/bundled/agents/python-code-auditor.md +49 -0
  18. package/src/gdskills/bundled/agents/refactoring-steward.md +61 -0
  19. package/src/gdskills/bundled/agents/security-auditor.md +62 -0
  20. package/src/gdskills/bundled/agents/test-first-driver.md +61 -0
  21. package/src/gdskills/bundled/agents/work-planner.md +62 -0
  22. package/src/gdskills/bundled/install-manifest.json +530 -0
  23. package/src/gdskills/bundled/rules/core/skill-lifecycle.mdc +29 -1
  24. package/src/gdskills/bundled/rules/core/skills-storage-workflow.mdc +2 -2
  25. package/src/gdskills/bundled/skills/review/code-style-review/SKILL.md +1 -1
  26. package/src/gdskills/bundled/skills/review/review-orchestrator/SKILL.md +74 -246
  27. package/src/gdskills/bundled/skills/review/review-orchestrator/output-contract.schema.json +19 -0
  28. package/src/gdskills/bundled/skills/review/review-orchestrator/reviewer-finding.schema.json +10 -0
  29. package/src/gdskills/bundled/skills/review/review-orchestrator/reviewer-input.schema.json +5 -0
  30. package/src/gdskills/bundled/skills/review/review-orchestrator/templates/pr-comment-backend.md +50 -0
  31. package/src/gdskills/bundled/skills/review/review-orchestrator/templates/pr-comment-frontend.md +52 -0
  32. package/src/gdskills/bundled/skills/review/review-orchestrator/templates/review-report.md +143 -0
  33. package/src/gdskills/bundled/stacks/go/agent-refs.json +3 -0
  34. package/src/gdskills/bundled/stacks/go/governance/eval.json +1745 -0
  35. package/src/gdskills/bundled/stacks/go/governance/scout.json +31 -0
  36. package/src/gdskills/bundled/stacks/go/pack.json +41 -0
  37. package/src/gdskills/bundled/stacks/go/rules/coding-style.mdc +85 -0
  38. package/src/gdskills/bundled/stacks/go/rules/patterns.mdc +65 -0
  39. package/src/gdskills/bundled/stacks/go/rules/security.mdc +73 -0
  40. package/src/gdskills/bundled/stacks/go/rules/testing.mdc +68 -0
  41. package/src/gdskills/bundled/stacks/go/skills/go-build-fix/SKILL.md +138 -0
  42. package/src/gdskills/bundled/stacks/go/skills/go-build-fix/evals.json +75 -0
  43. package/src/gdskills/bundled/stacks/go/skills/go-code-review/SKILL.md +121 -0
  44. package/src/gdskills/bundled/stacks/go/skills/go-code-review/evals.json +72 -0
  45. package/src/gdskills/bundled/stacks/go/skills/go-implementation/SKILL.md +122 -0
  46. package/src/gdskills/bundled/stacks/go/skills/go-implementation/evals.json +76 -0
  47. package/src/gdskills/bundled/stacks/go/skills/go-testing/SKILL.md +126 -0
  48. package/src/gdskills/bundled/stacks/go/skills/go-testing/evals.json +73 -0
  49. package/src/gdskills/bundled/stacks/python/agent-refs.json +3 -0
  50. package/src/gdskills/bundled/stacks/python/governance/eval.json +1758 -0
  51. package/src/gdskills/bundled/stacks/python/governance/scout.json +34 -0
  52. package/src/gdskills/bundled/stacks/python/pack.json +41 -0
  53. package/src/gdskills/bundled/stacks/python/rules/coding-style.mdc +63 -0
  54. package/src/gdskills/bundled/stacks/python/rules/patterns.mdc +88 -0
  55. package/src/gdskills/bundled/stacks/python/rules/security.mdc +84 -0
  56. package/src/gdskills/bundled/stacks/python/rules/testing.mdc +77 -0
  57. package/src/gdskills/bundled/stacks/python/skills/python-build-fix/SKILL.md +144 -0
  58. package/src/gdskills/bundled/stacks/python/skills/python-build-fix/evals.json +74 -0
  59. package/src/gdskills/bundled/stacks/python/skills/python-code-review/SKILL.md +155 -0
  60. package/src/gdskills/bundled/stacks/python/skills/python-code-review/evals.json +72 -0
  61. package/src/gdskills/bundled/stacks/python/skills/python-implementation/SKILL.md +143 -0
  62. package/src/gdskills/bundled/stacks/python/skills/python-implementation/evals.json +78 -0
  63. package/src/gdskills/bundled/stacks/python/skills/python-testing/SKILL.md +132 -0
  64. package/src/gdskills/bundled/stacks/python/skills/python-testing/evals.json +73 -0
  65. package/src/gdskills/bundled/stacks/react/agent-refs.json +4 -0
  66. package/src/gdskills/bundled/stacks/react/governance/eval.json +2188 -0
  67. package/src/gdskills/bundled/stacks/react/governance/scout.json +40 -0
  68. package/src/gdskills/bundled/stacks/react/pack.json +42 -0
  69. package/src/gdskills/bundled/stacks/react/rules/coding-style.mdc +58 -0
  70. package/src/gdskills/bundled/stacks/react/rules/patterns.mdc +79 -0
  71. package/src/gdskills/bundled/stacks/react/rules/security.mdc +70 -0
  72. package/src/gdskills/bundled/stacks/react/rules/testing.mdc +60 -0
  73. package/src/gdskills/bundled/stacks/react/skills/react-build-fix/SKILL.md +139 -0
  74. package/src/gdskills/bundled/stacks/react/skills/react-build-fix/evals.json +72 -0
  75. package/src/gdskills/bundled/stacks/react/skills/react-code-review/SKILL.md +148 -0
  76. package/src/gdskills/bundled/stacks/react/skills/react-code-review/evals.json +74 -0
  77. package/src/gdskills/bundled/stacks/react/skills/react-implementation/SKILL.md +140 -0
  78. package/src/gdskills/bundled/stacks/react/skills/react-implementation/evals.json +74 -0
  79. package/src/gdskills/bundled/stacks/react/skills/react-testing/SKILL.md +142 -0
  80. package/src/gdskills/bundled/stacks/react/skills/react-testing/evals.json +83 -0
  81. package/src/gdskills/bundled/stacks/react/skills/react-upgrade-migration/SKILL.md +155 -0
  82. package/src/gdskills/bundled/stacks/react/skills/react-upgrade-migration/evals.json +74 -0
  83. package/src/gdskills/bundled/stacks/ts-js-node/agent-refs.json +4 -0
  84. package/src/gdskills/bundled/stacks/ts-js-node/governance/eval.json +2155 -0
  85. package/src/gdskills/bundled/stacks/ts-js-node/governance/scout.json +40 -0
  86. package/src/gdskills/bundled/stacks/ts-js-node/pack.json +41 -0
  87. package/src/gdskills/bundled/stacks/ts-js-node/rules/coding-style.mdc +73 -0
  88. package/src/gdskills/bundled/stacks/ts-js-node/rules/patterns.mdc +61 -0
  89. package/src/gdskills/bundled/stacks/ts-js-node/rules/security.mdc +71 -0
  90. package/src/gdskills/bundled/stacks/ts-js-node/rules/testing.mdc +63 -0
  91. package/src/gdskills/bundled/stacks/ts-js-node/skills/nodejs-build-fix/SKILL.md +137 -0
  92. package/src/gdskills/bundled/stacks/ts-js-node/skills/nodejs-build-fix/evals.json +73 -0
  93. package/src/gdskills/bundled/stacks/ts-js-node/skills/nodejs-code-review/SKILL.md +124 -0
  94. package/src/gdskills/bundled/stacks/ts-js-node/skills/nodejs-code-review/evals.json +74 -0
  95. package/src/gdskills/bundled/stacks/ts-js-node/skills/nodejs-esm-migration/SKILL.md +152 -0
  96. package/src/gdskills/bundled/stacks/ts-js-node/skills/nodejs-esm-migration/evals.json +71 -0
  97. package/src/gdskills/bundled/stacks/ts-js-node/skills/nodejs-implementation/SKILL.md +127 -0
  98. package/src/gdskills/bundled/stacks/ts-js-node/skills/nodejs-implementation/evals.json +72 -0
  99. package/src/gdskills/bundled/stacks/ts-js-node/skills/nodejs-testing/SKILL.md +134 -0
  100. package/src/gdskills/bundled/stacks/ts-js-node/skills/nodejs-testing/evals.json +70 -0
@@ -0,0 +1,1758 @@
1
+ {
2
+ "schemaVersion": "1.0.0",
3
+ "reports": [
4
+ {
5
+ "schemaVersion": "1.0.0",
6
+ "skillId": "python/python-build-fix",
7
+ "strictness": "high",
8
+ "trials": 10,
9
+ "triggerAccuracy": {
10
+ "truePositive": 6,
11
+ "falsePositive": 0,
12
+ "positives": 6,
13
+ "negatives": 6
14
+ },
15
+ "evidence": "authored",
16
+ "scenarios": [
17
+ {
18
+ "id": "trigger-positive-1",
19
+ "kind": "trigger-positive",
20
+ "prompt": "Fix this ModuleNotFoundError: No module named 'mypkg.util'",
21
+ "strictness": "high",
22
+ "trials": 1,
23
+ "passes": 1,
24
+ "passRate": 1,
25
+ "passAtK": 1,
26
+ "grader": "trigger-rank-fork-family",
27
+ "status": "ran",
28
+ "deterministic": true
29
+ },
30
+ {
31
+ "id": "trigger-positive-2",
32
+ "kind": "trigger-positive",
33
+ "prompt": "Our editable pip install is broken, the package won't import",
34
+ "strictness": "high",
35
+ "trials": 1,
36
+ "passes": 1,
37
+ "passRate": 1,
38
+ "passAtK": 1,
39
+ "grader": "trigger-rank-fork-family",
40
+ "status": "ran",
41
+ "deterministic": true
42
+ },
43
+ {
44
+ "id": "trigger-positive-3",
45
+ "kind": "trigger-positive",
46
+ "prompt": "mypy is failing with a type error I don't understand, please fix it",
47
+ "strictness": "high",
48
+ "trials": 1,
49
+ "passes": 1,
50
+ "passRate": 1,
51
+ "passAtK": 1,
52
+ "grader": "trigger-rank-fork-family",
53
+ "status": "ran",
54
+ "deterministic": true
55
+ },
56
+ {
57
+ "id": "trigger-positive-4",
58
+ "kind": "trigger-positive",
59
+ "prompt": "ruff check is failing on our CI, fix the violations",
60
+ "strictness": "high",
61
+ "trials": 1,
62
+ "passes": 1,
63
+ "passRate": 1,
64
+ "passAtK": 1,
65
+ "grader": "trigger-rank-fork-family",
66
+ "status": "ran",
67
+ "deterministic": true
68
+ },
69
+ {
70
+ "id": "trigger-positive-5",
71
+ "kind": "trigger-positive",
72
+ "prompt": "pip is reporting a dependency resolver conflict between two packages",
73
+ "strictness": "high",
74
+ "trials": 1,
75
+ "passes": 1,
76
+ "passRate": 1,
77
+ "passAtK": 1,
78
+ "grader": "trigger-rank-fork-family",
79
+ "status": "ran",
80
+ "deterministic": true
81
+ },
82
+ {
83
+ "id": "trigger-positive-6",
84
+ "kind": "trigger-positive",
85
+ "prompt": "pytest collection is erroring out before any tests even run",
86
+ "strictness": "high",
87
+ "trials": 1,
88
+ "passes": 1,
89
+ "passRate": 1,
90
+ "passAtK": 1,
91
+ "grader": "trigger-rank-fork-family",
92
+ "status": "ran",
93
+ "deterministic": true
94
+ },
95
+ {
96
+ "id": "trigger-negative-1",
97
+ "kind": "trigger-negative",
98
+ "prompt": "Implement a new feature that uses asyncio.TaskGroup",
99
+ "strictness": "high",
100
+ "trials": 1,
101
+ "passes": 1,
102
+ "passRate": 1,
103
+ "passAtK": 1,
104
+ "grader": "trigger-rank-fork-family",
105
+ "status": "ran",
106
+ "deterministic": true
107
+ },
108
+ {
109
+ "id": "trigger-negative-2",
110
+ "kind": "trigger-negative",
111
+ "prompt": "Write pytest tests for the new widgets module",
112
+ "strictness": "high",
113
+ "trials": 1,
114
+ "passes": 1,
115
+ "passRate": 1,
116
+ "passAtK": 1,
117
+ "grader": "trigger-rank-fork-family",
118
+ "status": "ran",
119
+ "deterministic": true
120
+ },
121
+ {
122
+ "id": "trigger-negative-3",
123
+ "kind": "trigger-negative",
124
+ "prompt": "Review this Python diff for security issues",
125
+ "strictness": "high",
126
+ "trials": 1,
127
+ "passes": 1,
128
+ "passRate": 1,
129
+ "passAtK": 1,
130
+ "grader": "trigger-rank-fork-family",
131
+ "status": "ran",
132
+ "deterministic": true
133
+ },
134
+ {
135
+ "id": "trigger-negative-4",
136
+ "kind": "trigger-negative",
137
+ "prompt": "Fix the TypeScript build failure in our Node service",
138
+ "strictness": "high",
139
+ "trials": 1,
140
+ "passes": 1,
141
+ "passRate": 1,
142
+ "passAtK": 1,
143
+ "grader": "trigger-rank-fork-family",
144
+ "status": "ran",
145
+ "deterministic": true
146
+ },
147
+ {
148
+ "id": "trigger-negative-5",
149
+ "kind": "trigger-negative",
150
+ "prompt": "Run a full test-generation pass over this untested module",
151
+ "strictness": "high",
152
+ "trials": 1,
153
+ "passes": 1,
154
+ "passRate": 1,
155
+ "passAtK": 1,
156
+ "grader": "trigger-rank-fork-family",
157
+ "status": "ran",
158
+ "deterministic": true
159
+ },
160
+ {
161
+ "id": "trigger-negative-6",
162
+ "kind": "trigger-negative",
163
+ "prompt": "Review this code for architecture violations",
164
+ "strictness": "high",
165
+ "trials": 1,
166
+ "passes": 1,
167
+ "passRate": 1,
168
+ "passAtK": 1,
169
+ "grader": "trigger-rank-fork-family",
170
+ "status": "ran",
171
+ "deterministic": true
172
+ },
173
+ {
174
+ "id": "module-not-found-diagnosis",
175
+ "kind": "behavior",
176
+ "prompt": "Our Python project fails with `ModuleNotFoundError: No module named 'mypkg.util'` when running pytest. How do you diagnose and fix this?",
177
+ "strictness": "high",
178
+ "trials": 10,
179
+ "passes": 10,
180
+ "passRate": 1,
181
+ "passAtK": 1,
182
+ "grader": "judge",
183
+ "status": "ran",
184
+ "trialRecords": [
185
+ {
186
+ "output": "Diagnosis path for `ModuleNotFoundError: No module named 'mypkg.util'` under pytest:\n\n1. Reproduce and classify\n - Run the project's configured command exactly as CI would (`uv run pytest -x -q`, `poetry run pytest -x -q`, or plain `pytest -x -q`).\n - Read the *first* error. A `ModuleNotFoundError` during collection usually appears in a traceback that shows which file did the import \u2014 note that file and the exact dotted path (`mypkg.util`).\n - Distinguish two sub-cases: (a) the module genuinely doesn't exist at that path, or (b) the module exists but isn't importable from the test process's environment/sys.path. The error text is identical; the cause is not.\n\n2. Determine which case you're in\n - Does `src/mypkg/util.py` (or `mypkg/util.py` in a flat layout) actually exist? Check with a directory listing, not from memory.\n - If it does NOT exist: either the real module is named differently (e.g. `utils.py`), or it lives under a different package (e.g. `mypkg/parsing/util.py`), or it was moved/renamed and a caller kept the old path. Fix the import path or add the module \u2014 whichever reflects the intended structure.\n - If it DOES exist: it's a resolution problem (case b). Continue.\n - Confirm the active environment: `python -c \"import sys; print(sys.executable)\"` and `pip show mypkg` / `uv pip show mypkg`. If `mypkg` isn't installed, pytest is running against a source tree that isn't on `sys.path`.\n\n3. For case (b), the usual root causes\n - src-layout with no editable install: the package must be installed (`pip install -e .` / `uv pip install -e .`) or `pyproject.toml` must set `pythonpath = [\"src\"]` under `[tool.pytest.ini_options]`. Without one of these, `src/mypkg` is never on `sys.path`.\n - Broken editable install / package discovery: `[tool.setuptools.packages.find]` or `[tool.hatch.build]` excludes the package or points at the wrong directory, so `mypkg` installs without the submodule. Check the discovery config actually matches the real layout.\n - Missing `__init__.py`: if `mypkg/util.py` exists but `mypkg/__init__.py` (or a subpackage `__init__.py`) is absent, it may have accidentally become a namespace package that doesn't resolve the way the code assumes. Add the `__init__.py`.\n - Wrong environment active: a stale venv without the editable install, or a global interpreter. Reinstall into the active env or activate the right one.\n - Circular import: if `mypkg/__init__.py` imports `mypkg.util` and `mypkg.util` imports back into `mypkg`, you can get an import error during package init. The fix is to break the cycle (move the shared symbol, or import lazily inside the function) \u2014 not to wrap it in `try/except ImportError`.\n - `conftest.py` import error: if the failure is in `conftest.py`, the same diagnosis applies to that file's own imports and to pytest's rootdir/sys.path setup.\n\n4. Apply the smallest fix\n - Missing module / wrong path \u2192 correct the import or add the module at the intended location.\n - Not on sys.path \u2192 install editable, or configure pytest's `pythonpath`.\n - Broken discovery \u2192 fix the build config so `mypkg` installs with its submodules.\n - Missing `__init__.py` \u2192 add it.\n - Do NOT paper over it with `try/except ImportError: pass`, `--no-deps`, or a blanket `# type: ignore`/`# noqa`. Those hide the cause (a missing or mis-installed dependency) rather than fixing it.\n\n5. Verify\n - Re-run `pytest -x -q` \u2192 must exit 0, and the previously uncollectable tests must now actually run (not be skipped).\n - Re-run `ruff check .`, `ruff format --check .`, and `mypy .`/`pyright` to confirm no regressions.\n - `git status` should show only files touched by the real cause.\n\nReport format:\n\n```\nFixed: ModuleNotFoundError: No module named 'mypkg.util'\n Root cause: <e.g. src-layout project, mypkg not installed and pytest's\n pythonpath not set, so src/ was never on sys.path>\n Fix: <e.g. added pythonpath = [\"src\"] to [tool.pytest.ini_options]\n (or: pip install -e .)>\n Verified: ruff check, ruff format --check, mypy, pytest -x -q all green\n```\n\nThe key discipline: the message names a *symptom*. The first thing to establish is whether `mypkg/util.py` exists on disk (a path problem) or exists but is unreachable (an install/sys.path problem), because those two lead to entirely different fixes.",
187
+ "outputSha256": "ea02d26e359de4232f6f2d133bb62ea317b80fbb886184f48b24948d100bcd70",
188
+ "promptSha256": "3a31ea4e89f8542b980b797e019f04b07a0a5aefbdc325647b266486bf759333",
189
+ "deterministic": [],
190
+ "judge": {
191
+ "verdict": "pass",
192
+ "reason": "Step 2/3 identify concrete causes: module not installed / wrong env, src-layout path not on sys.path, circular import. Concrete fixes named: `pip install -e .`, `pythonpath = [\"src\"]` under [tool.pytest.ini_options], add __init__.py, break cycle. Step 5/report format distinguishes symptom (message) from root cause. No try/except suppression; it explicitly warns against it."
193
+ },
194
+ "passed": true
195
+ },
196
+ {
197
+ "output": "`ModuleNotFoundError: No module named 'mypkg.util'` inside pytest is almost always a collection/import problem, not a bug in your code. Diagnose it in this order:\n\n## 1. Find out what's actually on the import path\n\nRun pytest with verbose collection and, if you can, print the resolved package location:\n\n```bash\npython -c \"import mypkg; print(mypkg.__file__, mypkg.__path__)\"\npytest -x -q --collect-only\n```\n\nIf that `python -c` also fails, the problem is packaging/environment, not the test file. If it succeeds, the problem is specific to how pytest runs (e.g. rootdir/`src` layout, `conftest.py`, `sys.path`).\n\nAlso confirm which `mypkg` Python resolves \u2014 a stale copy installed in site-packages can shadow your source tree:\n\n```bash\npip show mypkg\npython -c \"import sys; print([p for p in sys.path])\"\n```\n\n## 2. Common root causes and their fixes\n\n**a) Flat vs `src/` layout mismatch (most common).**\nYou have `src/mypkg/util.py` but the package discovery config doesn't include it, or you're running pytest from a directory where `src/` isn't on `sys.path`. Check `pyproject.toml`:\n\n```toml\n[tool.setuptools.packages.find]\nwhere = [\"src\"]\ninclude = [\"mypkg*\"]\n```\n\nor for hatch:\n\n```toml\n[tool.hatch.build.targets.wheel]\npackages = [\"src/mypkg\"]\n```\n\nThen reinstall editable so the layout is picked up:\n\n```bash\npip install -e .\n```\n\nDon't add bare `sys.path` hacks in `conftest.py` as the fix \u2014 that hides a packaging misconfiguration.\n\n**b) Missing/renamed module.**\nConfirm the file is really named `util.py` (not `utils.py`), lives directly in `mypkg/`, and that `mypkg/__init__.py` exists (`mypkg/util.py` won't import as `mypkg.util` if `mypkg` is accidentally treated as a namespace package). Check the actual call site too \u2014 a typo like `from mypkg.utils import x` will produce exactly this error.\n\n**c) Editable install not linked.**\nIf you installed before creating `util.py`, the editable install may not have linked it. Re-run `pip install -e .`.\n\n**d) Shadowing.**\nA file literally named `util.py` at the repo root, or another `mypkg` on `sys.path`, can shadow the real one. Fix by removing the conflicting name \u2014 don't paper over it with `PYTHONPATH`.\n\n**e) Wrong environment.**\n`pytest` may be running under a different interpreter than the one where `mypkg` is installed. Check:\n\n```bash\nwhich python pytest\npython -m pytest -x -q # run pytest via the same interpreter\n```\n\nPrefer `python -m pytest` \u2014 it uses the active interpreter and puts the current dir on the path consistently.\n\n**f) Circular import.**\nIf `mypkg/__init__.py` imports something that imports `mypkg.util`, the error can present as `ModuleNotFoundError`. Move the shared symbol out of the cycle or import inside the function \u2014 do not wrap in `try/except ImportError`.\n\n## 3. Apply the smallest fix\n\nMatch the fix to the actual cause found above: correct the package-discovery config and reinstall editable, or fix the typo/name, or run under the right interpreter. Do not add `sys.path.insert(0, ...)` in `conftest.py` or a `try/except ImportError: pass` wrapper \u2014 both hide a real packaging/environment problem.\n\n## 4. Verify\n\nAll of these must exit 0:\n\n```bash\npython -c \"import mypkg.util; print(mypkg.util.__file__)\"\nruff check .\nmypy . # or pyright\npytest -x -q\n```\n\nRun the full `pytest -x -q`, not just collection \u2014 an import fix can change runtime behavior. Then check `git status`: only the config/source files whose cause you actually diagnosed should be touched, with no unrelated refactor.\n\n## Report format\n\n```\nFixed: ModuleNotFoundError: No module named 'mypkg.util'\n Root cause: src/mypkg/util.py existed but [tool.setuptools.packages.find]\n did not include \"mypkg*\" under where=[\"src\"], so the editable install\n never linked it into site-packages.\n Fix: added include=[\"mypkg*\"] to packages.find; re-ran `pip install -e .`\n Verified: import mypkg.util, ruff check, mypy, pytest -x -q all green\n```\n\nIf, after this, the only way to make it import is a `sys.path` hack or an editable reinstall that won't stick, say so explicitly in the report \u2014 that means the packaging config still needs a real fix rather than being resolved.",
198
+ "outputSha256": "a5eb915e6486e63b72394bed5986a55c9a47cadbf3e2276d3468c0500f40f7cd",
199
+ "promptSha256": "3a31ea4e89f8542b980b797e019f04b07a0a5aefbdc325647b266486bf759333",
200
+ "deterministic": [],
201
+ "judge": {
202
+ "verdict": "pass",
203
+ "reason": "Identifies concrete causes: src/flat layout mismatch, missing/renamed module, editable install not linked, shadowing, wrong environment, circular import. Names concrete fixes: pyproject [tool.setuptools.packages.find] where=[\"src\"] include=[\"mypkg*\"], `pip install -e .`, `python -m pytest`, fixing typo. Report format distinguishes root cause from symptom. Warns against try/except ImportError rather than using it."
204
+ },
205
+ "passed": true
206
+ },
207
+ {
208
+ "output": "`ModuleNotFoundError: No module named 'mypkg.util'` at pytest time means `mypkg` may import fine but the submodule `util` isn't resolveable \u2014 usually one of a few causes. Diagnose before fixing.\n\n## Step 1: Reproduce and get the exact failing import\n\n```bash\npytest -x -q # capture the FIRST traceback, not the whole dump\npython -c \"import mypkg; print(mypkg.__file__)\"\npython -c \"import mypkg.util; print(mypkg.util.__file__)\"\n```\n\nThe first test file's import line that raises is the real thing to chase \u2014 later errors are usually downstream of it.\n\n## Step 2: Classify what's actually wrong\n\nThere are four distinct causes for this exact error; the fix differs for each. Check them in order.\n\n**A. The file genuinely isn't there / is misnamed.**\n```bash\nls src/mypkg/ 2>/dev/null || ls mypkg/\n```\nIs there a `util.py`, or a `util/` directory with `__init__.py`? A common cause is the module being named `utils.py` (singular vs plural) or living under a different package. If it lives at a path the source doesn't expect, fix the import path to match reality \u2014 don't create a stub.\n\n**B. `mypkg` resolves to the wrong thing (wrong install / wrong environment).**\nIf `python -c \"import mypkg; print(mypkg.__file__)\"` points at a `site-packages` copy rather than your `src/mypkg`, your editable install is stale or points at an old layout. `mypkg` imports (an old copy without `util`), so the missing submodule is a symptom of a stale install:\n```bash\npip show mypkg # or: uv pip show mypkg\n```\nIf the recorded location is stale, reinstall editable: `pip install -e .` / `uv pip install -e .`.\n\n**C. Packaging config excludes the submodule.**\nEven with an editable install, if `pyproject.toml`'s package discovery doesn't include `mypkg.util`, it won't be linked. Check:\n- setuptools: `[tool.setuptools.packages.find]` `include`/`exclude`\n- hatch: `[tool.hatch.build.targets.wheel]` `packages`\n- src-layout vs flat layout \u2014 an editable install of a `src/`-layout project while config points at the flat root, or vice versa, produces exactly \"package imports but submodule doesn't.\"\n\n**D. Circular import.**\nIf `mypkg/__init__.py` or a sibling imports `util`, and `util` imports back into the package, you can get an import that fails only depending on import order. The traceback will show the cycle. The fix is a real restructure: move the shared symbol to a lower-level module, or do the import lazily inside the function \u2014 not a `try/except ImportError`.\n\n## Step 3: Apply the smallest fix for the cause you found\n\n- **A** \u2192 correct the import path / filename.\n- **B** \u2192 reinstall editable against the current layout.\n- **C** \u2192 add the submodule/package pattern to the discovery config, then reinstall editable.\n- **D** \u2192 restructure the cycle (move the symbol or import locally).\n\nDo not add `try/except ImportError: pass`, a stub `util.py`, or a `sys.path` hack \u2014 each hides the real cause and will resurface.\n\n## Step 4: Verify\n\n```bash\npython -c \"import mypkg.util; print(mypkg.util.__file__)\"\nruff check .\nmypy .\npytest -x -q\n```\nAll must exit 0. Run `pytest` even if you believe only imports changed \u2014 an install/layout fix can change what code actually executes at import time.\n\n## Step 5: Report the root cause, not the symptom\n\n```\nFixed: ModuleNotFoundError: No module named 'mypkg.util'\n Root cause: pyproject.toml's [tool.setuptools.packages.find].include omitted\n the util subpackage, so the editable install linked mypkg/ but not mypkg/util/.\n Fix: added \"mypkg.*\" to the include list and reinstalled editable.\n Verified: import mypkg.util, ruff, mypy, pytest -x -q all green.\n```\n\nThe single most important discipline here: `mypkg` importing while `mypkg.util` does not is diagnostic \u2014 it tells you the package itself is findable but a *specific submodule or the install layout* is broken. Chase that distinction rather than treating it as a generic \"module not found.\"",
209
+ "outputSha256": "e64083bc40b0de7658019a8a77ee1252206616e23516905b1af014fd9db3a161",
210
+ "promptSha256": "3a31ea4e89f8542b980b797e019f04b07a0a5aefbdc325647b266486bf759333",
211
+ "deterministic": [],
212
+ "judge": {
213
+ "verdict": "pass",
214
+ "reason": "Answer identifies concrete causes: stale/wrong editable install (B), packaging config excluding submodule (C), wrong import path (A), circular import (D). Names concrete fixes: `pip install -e .`, adding \"mypkg.*\" to [tool.setuptools.packages.find].include, correcting import path, restructuring cycle. Step 5 explicitly separates root cause from symptom. Warns against try/except ImportError rather than using it."
215
+ },
216
+ "passed": true
217
+ },
218
+ {
219
+ "output": "`ModuleNotFoundError: No module named 'mypkg.util'` on a pytest run means Python couldn't resolve the `mypkg.util` submodule at import time. The main suspects are: the package isn't installed in the active environment, `util` doesn't actually exist at the expected path, the package-discovery config excluded it, or there's a circular import. Walk through them in order.\n\n## Step 1: Reproduce with detail\n\nRun the project's own configured pytest command (discover the prefix from `pyproject.toml` \u2014 `uv run pytest -x -q`, `poetry run pytest -x -q`, or plain `pytest -x -q`). Read the *first* traceback, not the summary \u2014 the failing import is usually the root, and everything after it is downstream noise.\n\n## Step 2: Diagnose\n\n**a) Is `mypkg` installed in the active environment?**\n```bash\npip show mypkg # or: uv pip show mypkg\npython -c \"import mypkg; print(mypkg.__file__)\"\n```\nIf `mypkg` resolves but `mypkg.util` doesn't, the package is installed but incomplete \u2014 go to (c) or (d). If `mypkg` itself is missing, it's an install/environment problem: you're likely running pytest against a different interpreter than the one the project was installed into. Confirm with `which pytest` / `which python` matching the project venv.\n\n**b) Does `util` actually exist where the import expects it?**\n\nImports are case- and path-sensitive. Check `src/mypkg/util.py` (or `mypkg/util/__init__.py` for a subpackage). Common causes:\n- Typo in the import (`from mypkg.Util import ...` on a case-sensitive filesystem).\n- The file is named `utils.py` but code imports `util`.\n- The code imports `mypkg.util` but the real structure is `mypkg/parsers/util.py` \u2014 a wrong relative path.\n\n**c) Is this a `src/` layout with broken package-discovery config?**\n\nIf the repo uses `src/mypkg/` but `[tool.setuptools.packages.find]` (or `[tool.hatch.build]`) doesn't include `util`, the editable install links the package but omits the submodule \u2014 so `mypkg` imports fine and `mypkg.util` raises `ModuleNotFoundError`. This is the classic \"installed but incomplete\" signature.\n\nCheck `pyproject.toml` for the package discovery config. If it uses an explicit `include`/`packages` list, add the missing submodule (e.g. `include = [\"mypkg*\"]`). Also confirm `src/mypkg/util.py` exists and that `src/mypkg/__init__.py` is present (a missing `__init__.py` can make a directory a namespace package that the finder skips).\n\n**d) Circular import.**\n\nIf `util.py` exists and is included, but the error surfaces during import of `mypkg` (e.g. `mypkg/__init__.py` imports `mypkg.util`, which imports something back from `mypkg`), Python may report a partially-initialized module as `ModuleNotFoundError`. Look at the full traceback: if it shows the import chain looping back into `mypkg`, that's the real cause. Fix by moving the shared symbol to a lower-level module, or importing inside the function rather than at module top.\n\n**e) Pytest collection vs. runtime.**\n\nIf the traceback points at a *test file* or `conftest.py` importing `mypkg.util`, this is a collection error \u2014 the test never ran. The diagnosis is the same import-error path above; do not `pytest.mark.skip` the test to make collection pass.\n\n## Step 3: Apply the smallest fix\n\n- Wrong environment \u2192 activate the correct venv / run via the project's configured prefix; no code change.\n- Typo or wrong path \u2192 correct the import statement or rename the module to match, whichever is the actual intended name.\n- Discovery config omitted the submodule \u2192 add it to the package-find `include` list, then reinstall editable (`pip install -e .` / `uv pip install -e .`).\n- Circular import \u2192 restructure to break the cycle; do **not** wrap it in `try/except ImportError: pass`.\n- Missing `__init__.py` \u2192 add it if the directory is meant to be a regular package.\n\nTouch only what the failure requires \u2014 no unrelated refactor.\n\n## Step 4: Verify\n\nRe-run the full project toolchain, all must exit 0:\n```bash\nruff check .\nruff format --check .\nmypy . # or pyright\npytest -x -q\n```\nRun pytest even if the symptom was only an import error, since the fix can change runtime behavior.\n\n## Step 5: Report the root cause, not the symptom\n\n```\nFixed: ModuleNotFoundError: No module named 'mypkg.util'\n Root cause: src/mypkg/util.py existed but was excluded by\n [tool.setuptools.packages.find].include in pyproject.toml, so the\n editable install never linked the submodule.\n Fix: added \"mypkg*\" to include; reinstalled editable.\n Verified: ruff check, ruff format --check, mypy, pytest -x -q all green.\n```\n\n**Watch out for:** adding `# type: ignore`/`# noqa` (doesn't fix an import), pinning/downgrading with `--no-deps` to dodge a resolver conflict, wrapping the import in `try/except ImportError`, or disabling a rule project-wide. None fix the actual cause \u2014 a missing dependency or module must be declared/installed, or the import path corrected.",
220
+ "outputSha256": "2af038300caf73801470fe3a94746f3ef49936101104780342c66a96c7ba23db",
221
+ "promptSha256": "3a31ea4e89f8542b980b797e019f04b07a0a5aefbdc325647b266486bf759333",
222
+ "deterministic": [],
223
+ "judge": {
224
+ "verdict": "pass",
225
+ "reason": "Identifies concrete causes: not installed/wrong env (Step 2a, pip show/which python), wrong path (2b, src/mypkg/util.py vs utils.py), discovery config excluding submodule (2c), circular import (2d). Fixes named concretely: add \"mypkg*\" to include + pip install -e . (Step 3), correct import path, break cycle. Step 5 explicitly separates root cause from symptom. No try/except suppression; warns against it."
226
+ },
227
+ "passed": true
228
+ },
229
+ {
230
+ "output": "To diagnose and fix `ModuleNotFoundError: No module named 'mypkg.util'` under pytest, work through it in this order:\n\n**1. Reproduce and read the first error.**\nRun `pytest -x -q` and read the top of the output. The `ModuleNotFoundError` itself is the symptom; the full traceback tells you *who* is importing `mypkg.util` (a test file? `conftest.py`? `mypkg/__init__.py`? an already-imported module?). That caller is usually the real clue.\n\n**2. Check whether the module actually exists.**\n- Does `src/mypkg/util.py` (or `mypkg/util.py`) exist?\n- If instead there's a `src/mypkg/util/` directory, does it have an `__init__.py`? A directory without `__init__.py` won't be importable as `mypkg.util` in a regular (non-namespace) package.\n- Confirm the case is exact (`util.py` vs `Util.py`) \u2014 this bites silently on case-insensitive filesystems and fails on Linux/CI.\n\nIf the file is simply missing or misnamed, the fix is to create/rename it \u2014 that's the actual root cause, not an import-path issue.\n\n**3. Check the package layout vs. what's installed.**\nThis is the most common cause when the file *does* exist:\n- Is this a `src/`-layout project (`src/mypkg/...`) or a flat layout (`mypkg/...`)?\n- Run `pip show mypkg` (or `uv pip show mypkg`) \u2014 is the package installed at all in the *active* environment? `pytest` may be running against a different interpreter than the one where you installed things.\n- If it's installed editable, check that the build config actually includes the subpackage. For setuptools, `[tool.setuptools.packages.find]` with `where = [\"src\"]`; for hatch, `[tool.hatch.build.targets.wheel] packages = [\"src/mypkg\"]`. A common failure is that `mypkg` gets linked but `mypkg.util` is excluded by the discovery pattern, so the top-level package imports but its submodule doesn't.\n\n**4. Check for a circular import.**\nIf `mypkg/__init__.py` imports `mypkg.util`, and `mypkg/util.py` imports something back from `mypkg` at module top level, you can get an import error that surfaces as `ModuleNotFoundError`/`ImportError` during collection. The fix is a real restructure \u2014 move the shared symbol to a third module both can import, or defer one import into the function body \u2014 **not** a `try/except ImportError: pass` wrapper.\n\n**5. Check the pytest path configuration.**\nIf you rely on `rootdir`/`sys.path`, verify `pyproject.toml` or `pytest.ini` settings (`pythonpath = [\"src\"]` under `[tool.pytest.ini_options]`, or a `conftest.py` that adjusts `sys.path`). Running pytest from a different working directory than the config assumes can make `mypkg` resolvable but `mypkg.util` not, if only part of the tree is on the path.\n\n**6. Apply the smallest fix and verify.**\nFix whichever of the above is the actual cause \u2014 install/declare the missing dependency, add the missing `__init__.py`, correct the discovery config, or restructure the circular import. Then re-run:\n\n```\nruff check .\nruff format --check .\nmypy . # or pyright\npytest -x -q\n```\n\nAll must exit 0. Run the full test suite (not just the one failing file) since an import fix can change collection order and surface runtime regressions.\n\n**What not to do:** don't add `pytest.mark.skip` to the test file, don't wrap the import in `try/except ImportError: pass`, and don't blanket-suppress anything \u2014 a collection error means the test never runs at all, so hiding it just defers the problem.\n\n**Report format once fixed**, e.g.:\n\n```\nFixed: ModuleNotFoundError: No module named 'mypkg.util'\n Root cause: src/mypkg/util.py existed but pyproject.toml's\n [tool.setuptools.packages.find] include pattern omitted mypkg.util,\n so the editable install never linked it.\n Fix: tightened packages.find include; reinstalled editable.\n Verified: ruff check, mypy, pytest -x -q all green\n```\n\nIf the diagnosis points at a genuine version conflict or a larger structural problem you can't fix within scope, say so explicitly rather than papering over it with a pin or a suppression.",
231
+ "outputSha256": "c6d59915f9c8981c5f3b5432486cc243a4d10711bb65f7538f0a14e6c7648883",
232
+ "promptSha256": "3a31ea4e89f8542b980b797e019f04b07a0a5aefbdc325647b266486bf759333",
233
+ "deterministic": [],
234
+ "judge": {
235
+ "verdict": "pass",
236
+ "reason": "Identifies concrete causes: missing/misnamed file, wrong env (pip show mypkg), src/flat layout, packaging discovery excluding subpackage, circular import. Names concrete fixes: create/rename file, add __init__.py, setuptools packages.find where=[\"src\"], hatch packages=[\"src/mypkg\"], restructure circular import. Distinguishes symptom (traceback) from root cause and explicitly warns against try/except ImportError: pass."
237
+ },
238
+ "passed": true
239
+ },
240
+ {
241
+ "output": "I can't run anything in this exercise, so this is the diagnosis-and-fix plan you'd execute, not the executed result.\n\n## Diagnose\n\n`ModuleNotFoundError: No module named 'mypkg.util'` on pytest startup is almost always one of three things, in order of likelihood:\n\n1. **The module genuinely doesn't exist** \u2014 the file is named differently than the import expects (e.g. `utils.py` vs `util.py`, or it lives at `mypkg/core/util.py` but the import says `mypkg.util`), or it was never created/committed.\n2. **Editable install / package discovery doesn't include it** \u2014 `src/mypkg/util.py` exists on disk, but the `pyproject.toml` package-discovery config (`[tool.setuptools.packages.find]` or `[tool.hatch.build]`) doesn't pick up `mypkg`, or a missing `__init__.py` means it's treated as a namespace package that isn't installed. Then `import mypkg` works from some paths but `mypkg.util` doesn't resolve.\n3. **Wrong environment** \u2014 pytest is running against an interpreter where `mypkg` isn't installed at all, while a different venv has it. (If *no* `mypkg` import worked you'd more likely see failure on `mypkg` itself; seeing `mypkg` import fine and `mypkg.util` fail points at 1 or 2.)\n\nTo tell them apart:\n\n```bash\n# Which interpreter and where does mypkg actually resolve from?\npython -c \"import mypkg, sys; print(sys.executable); print(mypkg.__file__)\"\n\n# Does the module file actually exist on disk?\nls -la $(python -c \"import mypkg, os; print(os.path.dirname(mypkg.__file__))\")\n\n# Is the dist installed as editable, and what does its metadata say?\npip show mypkg # or: uv pip show mypkg\n```\n\n- If `mypkg.__file__` points into `site-packages` **without** the editable `.pth`/`__editable__` marker, and `util.py` only exists in your source tree, that's cause 2.\n- If `util.py` is simply absent or misnamed in the source tree, that's cause 1.\n- If `mypkg` itself won't even import, or `sys.executable` isn't the venv you expect, that's cause 3.\n\nAlso read the *first* traceback line \u2014 the collection error usually shows whether it's the test file importing `mypkg.util` or `mypkg/__init__.py` importing it internally (the latter is often a circular import that only manifests under pytest's import order).\n\n## Fix (smallest change for the actual cause)\n\n- **Cause 1 (missing/misnamed file):** rename or create `mypkg/util.py` to match the import, or fix the import path at the call site if the module really lives elsewhere. Don't shim it.\n- **Cause 2 (discovery/editable install):** add `mypkg*` to the discovery include list in `pyproject.toml`, add the missing `__init__.py`, then reinstall editable (`pip install -e .` / `uv pip install -e .`) and confirm `python -c \"import mypkg.util\"` succeeds from a path *outside* the project root.\n- **Cause 3 (wrong env):** install the project into the interpreter pytest is actually using (`python -m pytest` so pytest and the interpreter match), or activate the correct venv.\n- **If it's a circular import** surfaced by pytest: move the shared symbol to a lower-level module or import it lazily inside the function \u2014 never wrap it in `try/except ImportError`.\n\nDo **not** add `try/except ImportError: pass`, and do **not** `pytest.mark.skip` the test file \u2014 a collection error means the test never runs, and skipping hides that permanently.\n\n## Verify\n\n```bash\npython -c \"import mypkg.util\" # must succeed from outside the repo root\nruff check .\nruff format --check .\nmypy . # or pyright\npytest -x -q # all green, including the previously un-collectable file\n```\n\nReport format:\n\n```\nFixed: ModuleNotFoundError: No module named 'mypkg.util'\n Root cause: <e.g. util.py existed at src/mypkg/util.py but pyproject's\n package-find config excluded it, so the editable install never linked it>\n Fix: <e.g. added \"mypkg*\" to [tool.setuptools.packages.find].include; reinstalled -e>\n Verified: python -c \"import mypkg.util\", ruff check, mypy, pytest -x -q all green\n```\n\nName the root cause, not just the disappearing symptom, and confirm `git status` shows only the file(s) the diagnosis pointed at.",
242
+ "outputSha256": "feee5c6e006a944393b922bbb488b4757e8880f09bd5ba9c989016bfdaccabb9",
243
+ "promptSha256": "3a31ea4e89f8542b980b797e019f04b07a0a5aefbdc325647b266486bf759333",
244
+ "deterministic": [],
245
+ "judge": {
246
+ "verdict": "pass",
247
+ "reason": "Identifies concrete causes: missing/misnamed file, package-discovery/editable install excluding mypkg, wrong env, circular import. Fixes are concrete: rename/create mypkg/util.py, add 'mypkg*' to [tool.setuptools.packages.find].include + pip install -e ., install into pytest's interpreter, lazy import. Distinguishes root cause from symptom (report format, 'Name the root cause, not just the disappearing symptom'). Warns against try/except ImportError rather than using it."
248
+ },
249
+ "passed": true
250
+ },
251
+ {
252
+ "output": "`ModuleNotFoundError: No module named 'mypkg.util'` during pytest means: something imports `mypkg.util`, but that module isn't importable from the interpreter pytest is using. Diagnose in this order \u2014 the cause is almost always one of three things: wrong environment, a missing file, or a packaging/discovery config that excludes the module.\n\n## Step 1: Reproduce and pin down the exact importer\n\nRun the failing command directly to get the full traceback, not just the last line:\n\n```bash\npytest -x -q 2>&1 | head -40\n```\n\nThe `File \"...\", line ...` frames above the error tell you *which* file tried to import `mypkg.util` \u2014 could be the test file, `conftest.py`, or `mypkg/__init__.py`. That frame matters for Step 3.\n\nThen confirm which interpreter pytest is actually using:\n\n```bash\npython -c \"import sys; print(sys.executable)\"\npython -c \"import mypkg; print(mypkg.__file__)\"\n```\n\n## Step 2: Classify the cause\n\nWork through these checks; stop at the first one that's wrong.\n\n**A. Is the module file actually there?**\n\n```bash\nls src/mypkg/util.py src/mypkg/util/__init__.py mypkg/util.py 2>/dev/null\n```\n\nIf `util.py` (or `util/__init__.py`) doesn't exist, the import is simply targeting a module that was never created, renamed, or deleted \u2014 that's a plain typo/stale-reference bug. Grep for the importers and fix the path or create the missing module.\n\n**B. Is the *right* `mypkg` being imported?**\n\n`print(mypkg.__file__)` from Step 1 tells you. If it resolves to `site-packages/mypkg/...` from some *other* installed package rather than your working tree (common with `src/` layouts where the local package shadows nothing), the editable install is stale or pointed elsewhere. If it raises a different error entirely, the package isn't installed in this environment at all.\n\n**C. Is the environment the right one?**\n\n```bash\npip show mypkg # or: uv pip show mypkg\n```\n\nIf it's missing, the dependency/editable install never happened in this environment. If it's present but at a stale path, the editable install needs reinstalling.\n\n**D. Does the packaging config actually include `util`?**\n\nThis is the classic `src/`-layout trap and the most common real cause. Open `pyproject.toml` and check the discovery config:\n\n```toml\n[tool.setuptools.packages.find]\nwhere = [\"src\"]\ninclude = [\"mypkg*\"] # a bare \"mypkg\" here can exclude submodules\n```\n\nFor Hatch:\n\n```toml\n[tool.hatch.build.targets.wheel]\npackages = [\"src/mypkg\"] # must cover the subpackage\n```\n\nA common failure: layout is `src/mypkg/util.py` but the config only names the top-level package or lists subpackages explicitly and omits `util`. With an editable install that excludes it, `mypkg` imports fine while `mypkg.util` does not \u2014 exactly this error.\n\n**E. Is there a real circular import?**\n\nIf `mypkg/__init__.py` imports something that imports `mypkg.util`, and `util` in turn imports back into `mypkg`, you can get a partial-initialization `ModuleNotFoundError` that looks like a missing module. Check the traceback frames for a cycle (`mypkg/__init__.py` \u2192 `mypkg.util` \u2192 `mypkg/...`).\n\n## Step 3: Apply the smallest fix\n\n- **A (missing file):** create/rename the module, or fix the importer's path. Fix the stale reference at its source.\n- **B/C (stale or missing install):** reinstall the editable package:\n ```bash\n pip install -e . # or: uv pip install -e . / poetry install\n ```\n Verify only *after* fixing D if D is also wrong.\n- **D (discovery config excludes it):** correct the `include`/`packages` glob to cover the subpackage, then reinstall editable. Do **not** \"fix\" this by relying on implicit cwd-on-`sys.path` behavior \u2014 that makes tests pass locally while the installed package stays broken.\n- **E (circular import):** restructure \u2014 move the shared symbol to a lower-level module, or move the offending import inside the function that needs it. Do not wrap it in `try/except ImportError`.\n\nReinstall is required after any config change, since editable installs cache the package layout.\n\n## Step 4: Verify\n\n```bash\npython -c \"import mypkg.util; print(mypkg.util.__file__)\"\nruff check . && ruff format --check .\nmypy . # or pyright\npytest -x -q\n```\n\nAll must exit 0. Confirm `mypkg.util.__file__` points into your working tree, not a stale `site-packages` copy \u2014 that proves the editable install is now correct rather than the test just happening to import via cwd.\n\n## What *not* to do\n\n- Don't add a blank `__init__.py` in a random place to \"make it importable\" \u2014 under a `src/` layout with discovery config that ignores it, that changes nothing and hides the real config bug.\n- Don't `sys.path.append(...)` in `conftest.py`. It makes the test run pass while the package the user installs stays broken; fix the packaging config instead.\n- Don't `try/except ImportError` around the import.\n- Don't add a `# noqa`/`# type: ignore`; this is a runtime import error, and suppressing it doesn't make the module resolve.\n\n## Report format\n\n```\nFixed: ModuleNotFoundError: No module named 'mypkg.util'\n Root cause: src/mypkg/util.py existed, but pyproject.toml's\n [tool.setuptools.packages.find].include was \"mypkg\", excluding the\n subpackage, so the editable install linked mypkg but not mypkg.util.\n Fix: set include = [\"mypkg*\"], reinstalled with `pip install -e .`\n Verified: python -c \"import mypkg.util\" resolves to ./src/mypkg/util.py;\n ruff check, mypy, pytest -x -q all green\n```\n\nName the root cause in one sentence; if it turns out to be a genuine circular import or a deleted module rather than a config issue, say that explicitly instead \u2014 the symptom is identical but the fix is entirely different.",
253
+ "outputSha256": "aa3e7d61a7a76080364662ccd60c39b244c0665f2f2cde6a6045182c57bb521c",
254
+ "promptSha256": "3a31ea4e89f8542b980b797e019f04b07a0a5aefbdc325647b266486bf759333",
255
+ "deterministic": [],
256
+ "judge": {
257
+ "verdict": "pass",
258
+ "reason": "Answer identifies concrete causes (missing file, stale/missing editable install, src-layout discovery config excluding subpackage, circular import) and gives matched fixes: `pip install -e .`, corrected `include = [\"mypkg*\"]` / `packages = [\"src/mypkg\"]`, restructuring circular import. It distinguishes root cause from symptom (Step 2 classification, report format naming root cause) and explicitly warns against try/except ImportError rather than using it."
259
+ },
260
+ "passed": true
261
+ },
262
+ {
263
+ "output": "I can't actually run commands here, but here's how I'd diagnose and fix it end-to-end.\n\n## Diagnose\n\n1. Reproduce first: run `pytest -x -q` and read the *first* traceback line, not just the final summary. The `ModuleNotFoundError` is raised either by `mypkg` itself (a source module importing `mypkg.util`) or by a test/`conftest.py` doing so. The importing file's path tells you which.\n\n2. Check whether `mypkg.util` physically exists where Python would look for it:\n - Is there a `src/mypkg/util.py` or `src/mypkg/util/__init__.py` (src-layout) versus `mypkg/util.py` (flat layout)?\n - Confirm the actual file name \u2014 `utils.py` vs `util.py` is a frequent typo, and that's just a wrong import path in the importing module.\n\n3. Check the active environment. A very common cause is that pytest runs against an interpreter where `mypkg` was never installed or was installed without the file:\n - `python -c \"import mypkg; print(mypkg.__file__)\"` \u2014 if this points at an old editable install or site-packages copy, the environment is stale.\n - If a src-layout project wasn't installed editable at all, `mypkg` resolves only because you happened to run from the repo root; run pytest from another directory and the parent `mypkg` imports but its `util` submodule path can still be wrong.\n\n4. Distinguish the three real causes:\n - **Wrong import path** in the source (`import mypkg.utils` where the file is `util.py`). Symptom: `mypkg` imports fine, only `util` fails.\n - **File missing from the installed package** \u2014 often a packaging config that lists packages but not submodules, or a missing `__init__.py` in `mypkg/` so `util` isn't discovered. Check `[tool.setuptools.packages.find]` / `[tool.hatch.build]` / `[build-system]` in `pyproject.toml` actually covers `mypkg` and its subpackages.\n - **Stale editable install** \u2014 the file exists on disk but the installed metadata predates it. Reinstalling editable fixes it.\n\n5. If `mypkg` imports but `mypkg.util` raises `ModuleNotFoundError` *only inside a circular chain*, that's a circular import (`util` imports something from `mypkg.__init__` that imports `util`). Confirm by importing `mypkg.util` directly: `python -c \"import mypkg.util\"`.\n\n## Fix (smallest change for the actual cause)\n\n- **Typo / wrong path:** correct the import in the offending module (and in the test/`conftest.py` if that's where it is). Nothing else changes.\n- **Missing `__init__.py`:** add `mypkg/__init__.py` (or the `util/__init__.py` if it's meant to be a subpackage) so package discovery sees it.\n- **Packaging config:** add the package/glob to the build config's include list so the submodule ships, e.g. ensure `find` includes `mypkg*`.\n- **Stale env:** reinstall editable with the project's runner \u2014 `uv pip install -e .` / `pip install -e .` / `poetry install` \u2014 then re-run.\n- **Circular import:** move the shared symbol to a lower-level module (or import it lazily inside the function). Do **not** wrap it in `try/except ImportError: pass`, and do **not** add a blanket `# type: ignore`/`# noqa` \u2014 those hide the real problem.\n\n## Verify\n\nRe-run in order and require exit 0 on all:\n\n```\nruff check .\nruff format --check .\nmypy . # or pyright\npytest -x -q\n```\n\nThen check `git status` \u2014 only the file(s) whose actual cause you diagnosed should be modified, no unrelated refactor.\n\n## Report the cause, not the symptom\n\nState it like: \"Fixed `ModuleNotFoundError: No module named 'mypkg.util'`. Root cause: `src/mypkg/util.py` exists but the packaging `find` config didn't include `mypkg`, so the editable install never linked it. Fix: added `mypkg*` to the build config's include list, reinstalled editable. Verified: ruff, mypy, pytest all green.\" Naming the root cause is what lets the next person recognize the pattern.",
264
+ "outputSha256": "2b4887ee4eb9bad362f3536a992616c486b155edb8fdcbe92c100b28fe02b0bd",
265
+ "promptSha256": "3a31ea4e89f8542b980b797e019f04b07a0a5aefbdc325647b266486bf759333",
266
+ "deterministic": [],
267
+ "judge": {
268
+ "verdict": "pass",
269
+ "reason": "Answer identifies concrete causes (wrong import path typo utils vs util, missing __init__.py, packaging find config not covering mypkg, stale editable install, circular import) and names concrete fixes: correct import, add mypkg/__init__.py, add mypkg* to build config include list, `pip install -e .`/`uv pip install -e .`/`poetry install`, move shared symbol. It explicitly distinguishes root cause from symptom and warns against try/except ImportError: pass without committing it."
270
+ },
271
+ "passed": true
272
+ },
273
+ {
274
+ "output": "Diagnosing `ModuleNotFoundError: No module named 'mypkg.util'` under pytest \u2014 root cause and fix.\n\n## First: locate the actual failure\n\nThe name in the error tells you pytest is trying to import the submodule `mypkg.util`. The key question is *which* import is failing: a test file importing `mypkg.util`, or `mypkg`'s own `__init__.py` (or some module inside it) importing `mypkg.util`. Run:\n\n```bash\npytest -x -q # read the FIRST traceback, not later ones\npython -c \"import mypkg.util\"\n```\n\nIf `python -c \"import mypkg.util\"` also fails outside pytest, the problem is in the package/install, not pytest. If it succeeds outside pytest but fails inside, the problem is pytest's import path or rootdir configuration. That split determines everything below.\n\n## Step 1: Is the submodule actually installed / importable?\n\n```bash\npip show mypkg # or: uv pip show mypkg\npython -c \"import mypkg; print(mypkg.__file__)\"\n```\n\n- If `mypkg` isn't installed at all \u2192 the package is genuinely missing; install it (`pip install -e .` / `uv pip install -e .`).\n- If `mypkg` imports (prints a file path) but `mypkg.util` does not \u2192 the `util` source file isn't being picked up. Go to Step 2.\n- If neither imports \u2192 the environment/install itself is broken. Go to Step 3.\n\n## Step 2: The `util` module file isn't in the package\n\nCheck that the file actually exists where the package expects it:\n\n```bash\npython -c \"import mypkg, pathlib; print(pathlib.Path(mypkg.__file__).parent)\" # package dir\nls <that dir> # is util.py or util/ actually there?\n```\n\nCommon real causes:\n\n- **Missing `__init__.py` in a `util/` subpackage** \u2014 if `util` is a directory, and your build config treats it as a *namespace* package by mistake, or the directory simply lacks `__init__.py` under a build backend that requires it, it never gets linked. Add the `__init__.py`.\n- **Wrong layout vs. declared discovery.** If the project uses a `src/` layout, confirm the build config's package discovery actually includes the package. For setuptools, check `[tool.setuptools.packages.find]` (e.g. `where = [\"src\"]`, `include = [\"mypkg*\"]`); for hatch, check `[tool.hatch.build.targets.wheel] packages = [\"src/mypkg\"]`. A too-narrow `include` that omits `mypkg.util` (or a `find` pointed at the wrong `where`) is a classic cause: `mypkg` links, `mypkg.util` silently doesn't.\n- **Stale editable install.** If `mypkg/util.py` was added *after* the editable install, some backends don't pick up new files. Reinstall: `pip install -e . --force-reinstall` (or `uv pip install -e .`).\n\nFix the discovery config or add the missing file, then reinstall, and verify:\n\n```bash\npython -c \"import mypkg.util; print(mypkg.util.__file__)\"\n```\n\n## Step 3: Environment mismatch (installed elsewhere than pytest runs)\n\nIf `python -c \"import mypkg.util\"` works but pytest still fails, pytest is running under a *different* environment or a different `sys.path`. Check:\n\n```bash\nwhich pytest\npip show pytest mypkg # same env?\npytest -q --collect-only 2>&1 | head # does collection even start?\n```\n\nCommon fixes:\n\n- Invoke pytest through the project runner so it uses the same env: `uv run pytest`, or `poetry run pytest`, or `python -m pytest` instead of a bare `pytest`.\n- If there's a `conftest.py` at the repo root but pytest's rootdir/sys.path handling isn't adding the project root, add a minimal `pytest.ini`/`[tool.pytest.ini_options]` with `pythonpath = [\"src\"]` (or `[\".\"]` for a flat layout) so collection sees the package \u2014 but only after confirming the package is genuinely importable at all; don't use `pythonpath` to paper over a package that isn't installed.\n\n## Step 4: It's a circular import in disguise\n\nIf `mypkg.util` exists and is installed, but importing it fails only when reached *through* `mypkg`'s import chain, you may have a cycle: `mypkg/__init__.py` imports `mypkg.util`, which imports something that imports `mypkg` back, and Python raises `ModuleNotFoundError`/`ImportError` for the partially-initialized module. The fix is a restructure \u2014 move the shared symbol out, or import lazily inside the function \u2014 **not** a `try/except ImportError: pass` wrapper.\n\n## Apply the smallest fix\n\nWhatever Step 2\u20134 identified \u2014 add the missing `__init__.py`, widen the package-discovery `include`, reinstall the editable install, run pytest via the right interpreter, or break the cycle. Touch only that.\n\n## Verify\n\n```bash\npython -c \"import mypkg.util\" # exits 0\npytest -x -q # the original failing command now passes\nruff check . && mypy . # toolchain still green (catch regressions from the config change)\ngit status # only the files whose cause you diagnosed changed\n```\n\nYou're only done when the original pytest command exits 0 **and** `import mypkg.util` works from the same interpreter pytest uses.\n\n## Report format\n\n```\nFixed: ModuleNotFoundError: No module named 'mypkg.util'\n Root cause: pyproject.toml's [tool.setuptools.packages.find].include was\n [\"mypkg\"] (not \"mypkg*\"), so the mypkg.util subpackage was never linked\n by the editable install; mypkg itself still imported fine.\n Fix: changed include to [\"mypkg*\"] and reinstalled with `uv pip install -e .`\n Verified: `import mypkg.util` exits 0; ruff check, mypy, pytest -x -q all green\n```\n\nDon't add `# type: ignore`/`# noqa`, don't `try/except ImportError: pass` around the missing import, and don't `pytest.mark.skip` the test \u2014 a collection error means the test never runs at all, and skipping hides that permanently instead of fixing the import.",
275
+ "outputSha256": "eead9a6724cb640d21d45179221a6149dc427cd5c1e886e267345082741314e5",
276
+ "promptSha256": "3a31ea4e89f8542b980b797e019f04b07a0a5aefbdc325647b266486bf759333",
277
+ "deterministic": [],
278
+ "judge": {
279
+ "verdict": "pass",
280
+ "reason": "Identifies concrete causes: package not installed (pip install -e .), wrong src/flat layout discovery (setuptools packages.find include=[\"mypkg*\"]), env mismatch (uv run pytest), circular import. Names concrete fixes: changing include to [\"mypkg*\"], adding __init__.py, reinstalling. Distinguishes root cause from symptom in the report format. Explicitly warns against try/except ImportError: pass."
281
+ },
282
+ "passed": true
283
+ },
284
+ {
285
+ "output": "I can't run anything here, so treat this as the exact sequence to execute on your machine.\n\n## Diagnose\n\nFirst, pin down *which* of three very different causes this is, because the fix differs completely.\n\n**1. Does `mypkg/util.py` (or `mypkg/util/`) actually exist on disk?**\n```bash\nls mypkg/util.py mypkg/util/ 2>&1\n```\n- If it does not exist \u2192 it's a genuine missing file (typo'd module name, or a module that was renamed/deleted and some import still references `mypkg.util`). Find the referrers:\n ```bash\n grep -rn \"mypkg.util\\|from mypkg import util\" .\n ```\n and fix the import path or restore the module.\n\n**2. Does it exist, but the package isn't installed/linked?** Check layout and packaging config:\n```bash\nfind . -name util.py -o -name \"util\" -type d # where is it really?\ncat pyproject.toml # [build-system], package discovery\npython -c \"import mypkg; print(mypkg.__file__)\" # what got installed?\n```\nThe classic cause: a `src/` layout (`src/mypkg/util.py`) whose package-discovery config doesn't include it, so the editable install never linked it. Look at `[tool.setuptools.packages.find]` (`where = [\"src\"]`, `include = [\"mypkg*\"]`) or `[tool.hatch.build.targets.wheel]` (`packages = [\"src/mypkg\"]`). Also confirm `mypkg/__init__.py` exists \u2014 a missing `__init__.py` in a non-namespace-intended package makes discovery silently skip it.\n\n**3. Does it exist and install fine, but the import is circular?** If `mypkg/__init__.py` does `from mypkg.util import X` while `mypkg/util.py` does `from mypkg import Y`, you get a partial-import failure that surfaces as `No module named 'mypkg.util'`. Check `__init__.py`'s imports against `util.py`'s imports.\n\n## Fix (smallest change, per cause)\n\n- **Missing/renamed module** \u2192 correct the import path at the call site, or restore the module. Don't stub it.\n- **Packaging** \u2192 fix discovery so the real package dir is included, then reinstall:\n ```bash\n pip install -e . # or: uv pip install -e . / poetry install\n python -c \"import mypkg.util; print(mypkg.util.__file__)\"\n ```\n Add a missing `__init__.py` if that's the gap.\n- **Circular import** \u2192 move the shared symbol to a lower-level module that both can import, or import inside the function. Do not wrap in `try/except ImportError`.\n\nNote this is a *collection* error: pytest never ran the test, it failed importing it (or `conftest.py`). Fixing the import is the whole fix \u2014 do not `pytest.mark.skip` it.\n\n## Verify\n\n```bash\npython -c \"import mypkg.util\"\nruff check . && ruff format --check .\nmypy . # or pyright\npytest -x -q # must exit 0\n```\n\n`pytest` should now collect and run the affected tests. Run the full `pytest -x -q` even if you only touched packaging \u2014 an editable-install change can shift other imports.\n\n## Report format\n\n```\nFixed: ModuleNotFoundError: No module named 'mypkg.util'\n Root cause: src/mypkg/util.py existed but pyproject.toml's package-find\n config didn't include mypkg*, so the editable install never linked it.\n Fix: set [tool.setuptools.packages.find].where = [\"src\"] /\n include = [\"mypkg*\"], reinstalled with pip install -e .\n Verified: import OK, ruff, mypy, pytest -x -q all green\n```\n\nKeep the change to the diagnosed cause only \u2014 no unrelated refactor, no blanket `# noqa`/`# type: ignore`, and no pinning a dependency to route around an import problem.",
286
+ "outputSha256": "fc7ecb7d9fb391d11232a8958889f6bf0cfc6e424133842d481a9ab2391a66de",
287
+ "promptSha256": "3a31ea4e89f8542b980b797e019f04b07a0a5aefbdc325647b266486bf759333",
288
+ "deterministic": [],
289
+ "judge": {
290
+ "verdict": "pass",
291
+ "reason": "Identifies three concrete causes (missing file, src-layout packaging discovery, circular import). Names concrete fixes: corrected import path, `pip install -e .`, and specific config keys `[tool.setuptools.packages.find].where=[\"src\"]`/`include=[\"mypkg*\"]`. Report format separates root cause from symptom. Explicitly warns against try/except ImportError suppression."
292
+ },
293
+ "passed": true
294
+ }
295
+ ]
296
+ },
297
+ {
298
+ "id": "mypy-error-no-blanket-suppress",
299
+ "kind": "behavior",
300
+ "prompt": "mypy reports a type error on a function I touched. Fix it.",
301
+ "strictness": "high",
302
+ "trials": 10,
303
+ "passes": 10,
304
+ "passRate": 1,
305
+ "passAtK": 1,
306
+ "grader": "judge",
307
+ "status": "ran",
308
+ "trialRecords": [
309
+ {
310
+ "output": "I can't fix it from here \u2014 I have no tools, shell, or file access in this exercise, and you haven't shown me the error or the code. Guessing at a fix would be exactly the kind of suppression-and-move-on this skill is meant to prevent. Run it locally instead, following the `python-build-fix` workflow.\n\n**What to do, step by step:**\n\n1. **Get the exact error.** Run mypy scoped to the file and read the first error, not the later ones (they're often downstream of it):\n\n ```bash\n mypy path/to/file.py\n ```\n\n The message gives you the file, line, column, the actual inferred type, and the expected type. That mismatch *is* the bug to diagnose \u2014 don't skip past it.\n\n2. **Decide which side is actually wrong.** The error is one of:\n - **The call site is wrong** \u2014 you're passing the wrong type after the edit (e.g. `Optional[X]` where `X` is required). Fix the caller: guard the `None`, convert the value, or pass the right thing.\n - **The signature is wrong** \u2014 your edit changed the function's real contract, and the annotation no longer matches what it does. Fix the annotation to match the true behavior.\n - **The contract genuinely needs to accept more** \u2014 widen the annotation only if the function body actually handles the wider input. Don't widen just to silence the error.\n\n3. **Apply the smallest fix to the side that's actually wrong.** Since you touched this function, the most likely case is that the signature and the call sites drifted apart during your edit \u2014 reconcile them.\n\n**What not to do:**\n\n- Don't add `# type: ignore` to make it disappear. If the real fix is genuinely out of scope, leave the error *visible* and say so in your report rather than suppressing it.\n- Don't `cast(...)` your way around a real mismatch \u2014 that just relocates the lie.\n- Don't loosen the annotation to `Any` or broaden it past what the body handles.\n\n4. **Verify.** Re-run your project's configured commands and confirm all exit 0:\n\n ```bash\n ruff check .\n ruff format --check .\n mypy .\n pytest -x -q\n ```\n\n Run pytest even though it's only a type error \u2014 a signature change can introduce a runtime regression mypy won't catch.\n\n5. **Report the root cause, not the symptom**, e.g.:\n\n ```\n Fixed: mypy error at services/orders.py:142\n Root cause: signature annotated qty: int but the caller passes\n Optional[int] after my edit; body already guarded None.\n Fix: widened annotation to Optional[int] and added the None guard\n in the body (it was missing).\n Verified: ruff check, ruff format --check, mypy, pytest -x -q all green\n ```\n\nPaste the mypy output and the function you touched and I'll walk through the actual diagnosis with you.",
311
+ "outputSha256": "408cc19fc07f2203e9d1bffe982d5e42327d5d7650326fc0dcdcdd86bdb0945c",
312
+ "promptSha256": "a32dcc66b31d6b9b19df14395f5bb54ac669e4030ca35f7be4825e0413782228",
313
+ "deterministic": [],
314
+ "judge": {
315
+ "verdict": "pass",
316
+ "reason": "Answer asks for the exact mypy message and touched function ('Paste the mypy output and the function you touched'), and commits to fixing the real mismatch: call site, signature/annotation, or narrowing ('Fix the caller', 'Fix the annotation to match the true behavior'). It explicitly refuses # type: ignore, cast, and Any widening as substitutes. Satisfies all pass criteria; no fail criterion holds."
317
+ },
318
+ "passed": true
319
+ },
320
+ {
321
+ "output": "I can't fix it from here \u2014 I don't have access to your code, files, or a shell in this exercise. To actually resolve it, run the steps below in your environment.\n\n## What to do\n\n**1. Get the exact error, not a paraphrase**\n\n```bash\nmypy . # or: python -m mypy . / poetry run mypy .\n```\n\nMypy tells you three things you need: the file:line, the expected type, and the actual type it inferred. Paste that line back and I can give you a concrete fix. Reading the *first* error matters \u2014 later errors are often downstream of it.\n\n**2. Classify the mismatch**\n\nThe fix depends entirely on which of these it is:\n\n- **Your call site is wrong** \u2014 you're passing the wrong type into the function, or reading the wrong type out of it. Fix the call site.\n- **Your signature is wrong** \u2014 the function's annotation doesn't match what it actually does/returns. Fix the annotation to match reality.\n- **A narrowing/reveal issue** \u2014 mypy lost the type after a branch, `Optional` not narrowed, `dict.get()` returning `X | None`. Add an explicit check or `assert x is not None`, don't suppress.\n- **Third-party stub is missing/wrong** \u2014 the library ships no `py.typed` and mypy is guessing, or its stub is stale.\n\n**3. Fix the side that's actually wrong**\n\nDecide based on the function's real contract \u2014 its actual behavior and the intent of the code \u2014 not based on which edit makes the error disappear fastest. If the signature is a lie about what the function returns, the signature is the bug. If the caller is passing the wrong thing, the caller is the bug.\n\n**4. Verify**\n\n```bash\nmypy .\npytest -x -q # a type fix can still be a runtime regression\n```\n\n## Rules I'd hold myself to (and you should too)\n\n- **No blanket `# type: ignore`.** That hides the type hole permanently. If a real fix is genuinely out of scope, say so explicitly and leave the error visible rather than silently suppressing it.\n- **No `# type: ignore[code]` sprinkled to shut mypy up** unless there's an inline reason AND no real fix is available \u2014 and then the report must say why.\n- **Touch only what the failure requires.** Don't refactor unrelated code while fixing a type error.\n- **If the real cause is a stale/missing library stub**, that's the finding \u2014 say so, don't paper over it at the call site.\n\n## What I need from you\n\nIf you paste:\n\n1. The exact mypy output line (file:line + expected vs. actual), and\n2. The function signature and the failing call site,\n\nI'll give you the specific minimal edit and confirm it's the root cause rather than a symptom.",
322
+ "outputSha256": "074eb53543669bcdd5e52cb8176e9dcd2485252659631942ec1047bc1eeef56a",
323
+ "promptSha256": "a32dcc66b31d6b9b19df14395f5bb54ac669e4030ca35f7be4825e0413782228",
324
+ "deterministic": [],
325
+ "judge": {
326
+ "verdict": "pass",
327
+ "reason": "Answer asks for the exact mypy output line (file:line + expected vs actual) and the function signature/call site before proposing a fix (criterion 1). It commits to fixing the real side \u2014 call site, signature, or narrowing via assert x is not None (criterion 2). It explicitly refuses blanket # type: ignore and # type: ignore[code] suppression (criterion 3). No suppression or Any retyping is committed."
328
+ },
329
+ "passed": true
330
+ },
331
+ {
332
+ "output": "I don't have the actual error text, the file, or the function in front of me, so I can't produce a verified fix. What I can do is give you the exact procedure this skill prescribes, so you can apply it yourself \u2014 and I'll have to leave the \"verified\" boxes unchecked until you run it.\n\n**Step 1 \u2014 Reproduce and get the first error.**\nRun the project's own configured command. Find the run prefix from `pyproject.toml` (it may be `uv run mypy .`, `poetry run mypy .`, or plain `mypy .`). Read the **first** error only; later ones are frequently downstream of it.\n\n**Step 2 \u2014 Classify and locate the root cause.**\nA mypy error on a function you just touched is almost always one of:\n\n- **Signature changed, call sites not updated** \u2014 you altered a parameter/return type and something still passes the old shape. Fix the call site, or decide the *signature* is what's wrong.\n- **Inferred type narrowed or widened unexpectedly** \u2014 e.g. a default of `None` now makes the parameter `Optional[X]` where the body assumes `X`.\n- **Missing/incomplete annotation** \u2014 the function is untyped and mypy infers something too narrow. Add the real annotation; do not sprinkle `Any`.\n- **Third-party stubs missing or wrong** \u2014 a library ships no types and mypy falls back to `Any`/errors at the boundary.\n\nThe key decision, straight from the skill: **fix whichever of the two is actually wrong given the function's real contract \u2014 not whichever silences the error fastest.** If the caller is passing the wrong type, fix the caller. If the function's declared signature is wrong, fix the signature.\n\n**Step 3 \u2014 Apply the smallest fix.**\nOnly change what the failure requires. Don't refactor the surrounding module while you're in there.\n\n**Step 4 \u2014 Verify.**\nRe-run, in order:\n- `mypy .` (or `pyright`)\n- `ruff check .`\n- `ruff format --check .`\n- `pytest -x -q`\n\nThat last one matters even though the original failure was a type error \u2014 a signature change can introduce a runtime regression mypy never sees.\n\n**Constraints you must not violate (from the skill's Rules):**\n- Do **not** add `# type: ignore` to make the error disappear. If the real fix is genuinely out of scope, say so explicitly in your report and leave the error visible rather than silently suppressing it.\n- Do **not** edit mypy config or `pyproject.toml` to exclude the file or disable the check as a way of \"fixing\" it.\n- Do **not** widen everything to `Any` to appease the checker.\n\n**Red flag to watch for:** the rationalization \"I'll just add `# type: ignore`, the real fix is bigger.\" That hides the type hole permanently. If a proper fix truly isn't available in scope, report that \u2014 don't suppress.\n\n**What I need from you to do this concretely:** the exact mypy output line, the function (signature + body) you changed, and the call sites. Paste those and I'll give you the specific smallest fix and the report block in the skill's format:\n\n```\nFixed: <the mypy error>\n Root cause: <signature vs call-site vs annotation vs stubs>\n Fix: <the one-line change>\n Verified: mypy, ruff check, ruff format --check, pytest -x -q all green\n```\n\nI can't mark that \"Verified\" line honestly until those four commands exit 0 on your machine \u2014 I have no tools here, so the verification step is yours to run.",
333
+ "outputSha256": "856bf9c46f2c3a580c704d543218e5bfcd617f3c8b81e1ae99a181fcfe2e642b",
334
+ "promptSha256": "a32dcc66b31d6b9b19df14395f5bb54ac669e4030ca35f7be4825e0413782228",
335
+ "deterministic": [],
336
+ "judge": {
337
+ "verdict": "pass",
338
+ "reason": "Answer asks for the exact mypy output line, the touched function (signature+body), and call sites before proposing a fix (Step 1, 'What I need from you'). It commits to fixing the real mismatch: 'fix whichever of the two is actually wrong... If the caller is passing the wrong type, fix the caller. If the function's declared signature is wrong, fix the signature.' It explicitly refuses # type: ignore, config edits, and widening to Any."
339
+ },
340
+ "passed": true
341
+ },
342
+ {
343
+ "output": "I can't actually fix it from here \u2014 I have no file or shell access in this exercise, so I don't know which function, which error, or what your project looks like. What I can do is tell you exactly how to fix it correctly once you share the mypy output, and give you the diagnostic path.\n\n## What I need from you\n\nPaste the actual mypy error. It looks like:\n\n```\nsrc/mypkg/foo.py:42: error: Argument 1 to \"bar\" has incompatible type \"str | None\"; expected \"str\" [arg-type]\n```\n\nThe exact file:line, the types mypy says don't match, and the error code matter \u2014 \"incompatible type\" and \"returning Any\" and \"has no attribute\" are three different fixes.\n\nAlso useful: the function signature and its call site(s). A type error is almost never actually *at* the reported line \u2014 it's at the boundary between the function's declared contract and whatever is passing values across it.\n\n## How to fix it (don't suppress it)\n\nThe fast wrong move is `# type: ignore` on the reported line. That hides the hole permanently and leaves the next reader with a lie. The right move depends on which side is actually wrong:\n\n**1. The call site passes a value the function never promised to accept.**\nThe signature is correct; the caller is wrong. Fix the caller \u2014 narrow the value, handle the `None` case explicitly, or convert the type. Example: mypy says `\"str | None\"` passed where `\"str\"` is expected \u2014 add the `if x is None: ...` guard or raise, don't widen the parameter to `str | None` unless the function genuinely needs to handle it.\n\n**2. The signature is too narrow for what the function actually does.**\nThe caller is right; the annotation is stale. Widen/adjust the annotation to reflect the real contract. This is the correct fix when the function already handles the broader input \u2014 you're just documenting reality.\n\n**3. The value's type is a real bug, not just a mypy complaint.**\nIf mypy says you're calling a method on something that might be `None`, that may be a genuine `AttributeError` waiting to happen at runtime. Fix the logic, not the annotation.\n\nTo decide between 1 and 2, look at every call site. If they all pass values consistent with one interpretation, that's your contract. If one caller is the outlier, that caller is usually the bug.\n\n**4. Third-party library is untyped or wrongly typed.**\n`types-<lib>` stub package, or a targeted `cast()`, or an inline `# type: ignore[import-untyped]` with a reason. Justify it \u2014 this is the one case where a localized suppression can be legitimate, but say why in the report.\n\n## Verify\n\nAfter the fix, re-run:\n\n```bash\nmypy .\nruff check .\nruff format --check .\npytest -x -q\n```\n\n`pytest` matters even for a type-only fix \u2014 a signature change can regress runtime behavior that the type checker won't catch. All four must exit 0. If you added no suppression, `git diff` should show only the function and its call sites, nothing unrelated.\n\n## What I'd write in the report\n\nOnce you paste the error and the signature, I'll give you the concrete one-line fix and the report in this form:\n\n```\nFixed: src/mypkg/foo.py:42 arg-type: \"str | None\" passed where \"str\" expected\n Root cause: parse_config() annotation narrowed to str in a refactor, but\n load_user_prefs() can return None when the prefs file is absent.\n Fix: handle None at the load_user_prefs() call site (raise on missing prefs)\n rather than widening parse_config's parameter.\n Verified: mypy, ruff check, ruff format --check, pytest -x -q all green\n```\n\nPaste the mypy output and the function, and I'll turn that into the actual fix.",
344
+ "outputSha256": "e6f65abb7d3faedfd83868e38c50ca074903d07dd1c51672913ec6670a26e99d",
345
+ "promptSha256": "a32dcc66b31d6b9b19df14395f5bb54ac669e4030ca35f7be4825e0413782228",
346
+ "deterministic": [],
347
+ "judge": {
348
+ "verdict": "pass",
349
+ "reason": "Asks for the actual mypy message and touched function/signature before fixing (\"Paste the actual mypy error\", \"the function signature and its call site(s)\"). Commits to fixing the real mismatch: narrow the caller, adjust the annotation, or fix logic. Explicitly refuses suppression: \"The fast wrong move is `# type: ignore`... hides the hole permanently.\" No blanket ignore/Any/config loosening as the fix."
350
+ },
351
+ "passed": true
352
+ },
353
+ {
354
+ "output": "I can't fix it from here \u2014 I have no shell or file access in this exercise, and more importantly I don't have the actual mypy output or the code. A \"type error on a function I touched\" isn't enough to fix correctly, because the right fix depends entirely on *which side* is wrong.\n\nHere's how to resolve it yourself, following the build-fix workflow:\n\n**Step 1 \u2014 Get the exact error.** Run the project's configured command (`mypy .` or `pyright`, or via `uv run` / `poetry run`). Read the *first* error, not the ones after it \u2014 later errors are often downstream. Copy the exact line, e.g.:\n\n```\nsrc/mypkg/parsers.py:42: error: Argument 1 to \"parse\" has incompatible type \"str | None\"; expected \"str\" [arg-type]\n```\n\n**Step 2 \u2014 Decide which side is actually wrong.** The error means mypy found a mismatch between what a call site passes/provides and what the function signature declares. One of the two is the real bug:\n\n- **The signature is wrong** \u2014 the function genuinely accepts `None` (or a wider type) and should say so: widen the parameter, e.g. `def parse(text: str | None) -> ...`, and handle `None` in the body.\n- **The call site is wrong** \u2014 the function really does require a `str`, and the caller is passing a value that might be `None`. Fix the caller, e.g. guard with `if text is None: ...`, use a default, or narrow before the call.\n- **The two are fine but mypy can't see it** \u2014 e.g. a re-export or a genuinely well-typed but opaque third-party API. This is the *only* case where a targeted `# type: ignore[<specific-code>]` is defensible, and it must have an inline reason.\n\nPick based on the function's real contract \u2014 whichever silences it fastest is the wrong criterion.\n\n**Step 3 \u2014 Apply the smallest fix.** Change the signature or the one call site. Don't refactor surrounding code while fixing a build failure.\n\n**Step 4 \u2014 Verify.** Re-run `mypy .` *and* `pytest -x -q` (a signature change can break runtime behavior mypy won't catch), plus `ruff check .` and `ruff format --check .`. All must exit 0.\n\n**Rules I'd hold you to here:**\n- Do **not** add a blanket `# type: ignore` just to make it disappear. If the real fix is genuinely out of scope, say so explicitly and leave the error visible rather than silently suppressing it.\n- If you do suppress, use a *specific* code (`# type: ignore[arg-type]`, not bare `# type: ignore`) with an inline reason.\n- `git status` should show only the file(s) involved in the actual cause.\n\n**Report format** once green:\n\n```\nFixed: mypy arg-type error in src/mypkg/parsers.py\n Root cause: parse() was annotated to take `str`, but callers can pass\n None when the content-type header is missing.\n Fix: widened the parameter to `str | None` and handled None in the body.\n Verified: ruff check, ruff format --check, mypy, pytest -x -q all green\n```\n\nIf you paste the exact mypy line and the function (signature + the flagged call site), I'll tell you which side is wrong and the precise edit.",
355
+ "outputSha256": "98b76d693804fc3d2b8af6f53565d6e6ce197f682d0f91b396c1e51ddb8c889c",
356
+ "promptSha256": "a32dcc66b31d6b9b19df14395f5bb54ac669e4030ca35f7be4825e0413782228",
357
+ "deterministic": [],
358
+ "judge": {
359
+ "verdict": "pass",
360
+ "reason": "Asks for the exact mypy line and the function (signature + flagged call site) before proposing a fix, since neither was given. Commits to fixing the real mismatch: widen the annotation, fix the call site, or narrow. Explicitly refuses blanket `# type: ignore`/`# noqa` and warns against silencing fastest. No suppression or Any used as the fix."
361
+ },
362
+ "passed": true
363
+ },
364
+ {
365
+ "output": "I can't fix it without seeing the actual code and the actual error. \"A type error on a function I touched\" isn't enough to diagnose, and I have no file access in this exercise \u2014 so let me give you the exact procedure and what I'd need to see.\n\n**What I need from you:**\n1. The full mypy error text, including file, line, column, and error code (e.g. `error: Incompatible return value type (got \"str | None\", expected \"str\") [return-value]`).\n2. The function's signature and body, plus any call sites mypy flags.\n3. The relevant type definitions (the types flowing in and out \u2014 dataclass/TypedDict/Protocol, whatever the parameter and return types are).\n\n**How I'd resolve it, once I have that:**\n\n**Step 1 \u2014 Read the exact mismatch.** mypy names the two types that collide and where. Don't skim past it; the fix follows from which side is actually wrong.\n\n**Step 2 \u2014 Find the root cause, not the silence.** A type error on a function you touched almost always means one of:\n- The signature drifted from the real contract (you changed the body so it can now return `None`/a union, or accept a wider input).\n- A call site passes the wrong shape.\n- A dependency's stub is stricter than its runtime behavior.\n\nDecide which side is *actually* wrong given the function's real contract. Fix that side:\n- If the body genuinely can return `None` but callers can't handle it, the signature should say `-> str | None` and you handle the `None` at the call sites \u2014 or you make the body guarantee non-`None`. Pick based on the real contract.\n- If the call site is wrong, fix the call site.\n- If you're narrowing a union, use an actual `isinstance`/`assert x is not None` guard that reflects a real invariant, not a bare `cast` that lies.\n\n**Step 3 \u2014 Apply the smallest change** that fixes the cause, touching only that function and its immediate call sites.\n\n**Step 4 \u2014 Verify:** re-run `mypy .` (plus the project's ruff/format/pytest \u2014 a type fix can hide a runtime regression), all must exit 0.\n\n**Rules I will not bend on:**\n- No `# type: ignore` / `# noqa` to make a real error disappear. If the proper fix is genuinely out of scope, I'll say so in the report and leave the error visible rather than suppress it.\n- No `cast()` papering over a mismatch the code doesn't actually guarantee.\n- Report the root cause, not just the symptom.\n\nPaste the error and the function, and I'll give you the concrete diff.",
366
+ "outputSha256": "6be0d342da376d150b1df5342010bc291f3bc6264ca6d5801e3b5e40bdb89b0c",
367
+ "promptSha256": "a32dcc66b31d6b9b19df14395f5bb54ac669e4030ca35f7be4825e0413782228",
368
+ "deterministic": [],
369
+ "judge": {
370
+ "verdict": "pass",
371
+ "reason": "Asks for the exact mypy message and touched function (\"What I need from you: 1. The full mypy error text... 2. The function's signature and body\"), commits to fixing the real mismatch (signature/call site/narrowing via isinstance/assert), and explicitly refuses suppression: \"No # type: ignore / # noqa to make a real error disappear.\" No fail criterion met."
372
+ },
373
+ "passed": true
374
+ },
375
+ {
376
+ "output": "I can't fix it from here \u2014 I don't have the mypy output, the source, or any file access. But here's exactly what to do, following the smallest-root-cause path:\n\n**1. Read the actual error, not the summary.**\nRun `mypy .` and read the first error in full. Mypy errors have a shape that tells you the cause:\n\n```\nsrc/foo.py:42: error: Argument 1 to \"parse\" has incompatible type \"str | None\"; expected \"str\" [arg-type]\n```\n\nThe `[arg-type]` (or `[return-value]`, `[assignment]`, `[attr-defined]`, `[call-overload]`, etc.) code tells you *what kind* of mismatch it is. Note file, line, and the exact expected-vs-actual types.\n\n**2. Find which side is actually wrong.**\nThis is the whole job \u2014 don't guess:\n\n- **Call site is wrong**: you passed a value that can genuinely be `None`/`int`/whatever and the callee legitimately requires `str`. Fix the call site (narrow before the call, or handle the `None` case).\n- **Signature is wrong**: the callee's annotation is too narrow for what it really accepts and does. Fix the annotation \u2014 don't loosen it to `Any`.\n- Never pick whichever side silences mypy fastest; pick the side that matches the function's *real* contract.\n\n**3. Common cases from \"I touched this function\":**\n\n- You added a parameter or changed a type and haven't updated callers \u2192 fix the callers.\n- A value from a dict/`os.environ`/JSON is `str | None` and you now pass it where `str` is required \u2192 narrow it (`if x is None: raise ...`) before passing.\n- You changed a return type and callers still assume the old one \u2192 update the callers.\n- Generic/`Optional` inference drift \u2192 add the explicit annotation, don't cast.\n\n**4. Forbidden shortcuts** (these are the ones this skill blocks):\n- `# type: ignore` to make it vanish without an inline justification.\n- Weakening an annotation to `Any`.\n- `cast(...)` purely to silence the checker.\n\n**5. Verify:**\n- `mypy .` exits 0.\n- `pytest -x -q` still passes (a type fix can change runtime behavior).\n- `git status` shows only the file(s) whose cause you diagnosed \u2014 no unrelated refactor.\n\nPaste the mypy output and the function (and its callers if the error is at a call site) and I'll give you the exact smallest diff.",
377
+ "outputSha256": "30ac479d3f2f01afe36159260cabdcbe4d12057eb06e417f269d2629efbc9fd3",
378
+ "promptSha256": "a32dcc66b31d6b9b19df14395f5bb54ac669e4030ca35f7be4825e0413782228",
379
+ "deterministic": [],
380
+ "judge": {
381
+ "verdict": "pass",
382
+ "reason": "Answer asks for the mypy output and the touched function/callers before proposing a fix (step 1, closing line 'Paste the mypy output and the function...'), and commits to fixing the real mismatch: 'Fix the call site (narrow before the call)' or 'Fix the annotation \u2014 don't loosen it to Any'. Step 4 explicitly refuses '# type: ignore', Any, and cast as fixes."
383
+ },
384
+ "passed": true
385
+ },
386
+ {
387
+ "output": "I can fix the type error, but I need more than \"a type error\" to give you a real fix rather than a guess. Here's what I can tell you from the workflow, plus what I'd need from you.\n\nThe fix depends on the actual mismatch mypy reported. There are three common shapes, and each has a different correct resolution:\n\n**1. The call site is wrong.** You're passing the wrong type. Fix the caller \u2014 coerce, narrow, or pass the right value:\n```python\ndef process(items: list[str]) -> None: ...\n\nprocess(\"abc\") # error: Argument 1 has incompatible type \"str\"\nprocess([\"a\", \"b\"]) # correct fix \u2014 the caller was wrong\n```\n\n**2. The signature is wrong.** The function never actually required what its annotation claims, and the body proves it. Fix the annotation:\n```python\ndef count(x: int) -> int: # error at the call site: \"str\" not \"int\"\n return len(x) # ...but len() only accepts sized things\ndef count(x: str) -> int: # correct fix \u2014 the annotation was wrong\n```\n\n**3. Optionality or None-handling.** The value can be `None` and you're using it as if it can't:\n```python\ndef find(name: str) -> Item | None: ...\n\nitem = find(\"x\")\nprint(item.name) # error: \"None\" has no attribute \"name\"\nif item is None:\n raise KeyError(\"x\") # correct fix \u2014 handle it, don't suppress it\nprint(item.name)\n```\n\n**What I will not do:**\n\n- Add `# type: ignore` to silence it. If the real fix is genuinely out of scope for your change, the honest move is to leave the error visible and say in the commit/report that a larger fix is needed \u2014 not to hide the hole permanently.\n- Add an `assert isinstance(...)` purely to satisfy mypy when the runtime type isn't actually guaranteed. That's a lie to the type checker that becomes a real bug when the assert is optimized out under `-O`.\n- Refactor unrelated code around the function to make the error go away. The fix should be confined to the actual cause.\n\n**What I need from you to apply the right one:**\n\n1. The exact mypy output \u2014 file, line, and the full message text (e.g. `error: Incompatible types in assignment (expression has type \"str\", variable has type \"int\")`).\n2. The function definition and the call site(s) involved, or at least their signatures.\n3. Whether the function's annotation or the caller is the thing you consider authoritative \u2014 i.e. did you change what the function *should* accept, or did you change a caller assuming the old contract held?\n\nThat distinction is the whole decision. If you touched the function's contract, the callers may now be wrong and mypy is correctly flagging them. If you only touched the body, the annotation may no longer match reality.\n\n**Verification once fixed:** I'd re-run `mypy .` and confirm the specific error is gone and no new ones appeared, then run `ruff check .` and `pytest -x -q` \u2014 a signature change can break callers mypy catches but a runtime-only change can slip past the checker entirely, so the test run is not optional. I'd report the root cause (which of the three shapes it was, and whether the annotation or the caller was wrong), not just \"fixed the type error.\"",
388
+ "outputSha256": "3cea496fe74b3799503a033178a05c5d0f44f4e998612bd1f651a260d4c28ac3",
389
+ "promptSha256": "a32dcc66b31d6b9b19df14395f5bb54ac669e4030ca35f7be4825e0413782228",
390
+ "deterministic": [],
391
+ "judge": {
392
+ "verdict": "pass",
393
+ "reason": "Answer asks for the exact mypy message and touched function/call site before fixing (criterion 1), commits to fixing the real mismatch via caller, annotation, or None-narrowing (criterion 2), and explicitly refuses # type: ignore and other suppressions (criterion 3). No fail criterion: it never adds a suppression or Any."
394
+ },
395
+ "passed": true
396
+ },
397
+ {
398
+ "output": "I don't have the error text or the code in front of me, so I can't name the actual fix \u2014 but here's the diagnosis-and-fix procedure, and what I need from you to make the change precise.\n\n**What I need**\n- The exact mypy output (error code + message + file:line), e.g. `error: Argument 1 to \"parse\" has incompatible type \"str | None\"; expected \"str\" [arg-type]`.\n- The function signature and the call site (or the body line) mypy flags.\n\n**Diagnosis (Step 2)**\n\nThe message itself tells you which side is wrong, but you have to decide it against the function's *real contract*, not by whatever silences the error:\n\n- **`arg-type`** at a call site \u2014 either the caller is passing something genuinely invalid (fix the call site: narrow with a guard, convert the type), or the parameter is typed too narrowly for its real use (widen the annotation). Pick based on what the function actually accepts.\n- **`return-value`** inside the function \u2014 decide whether the return annotation is wrong or the returned expression is. Don't cast to force a match.\n- **`Optional`/`None` mismatch** (`str | None` vs `str`) \u2014 if the value can truly be `None`, the fix is a real guard or an early return at the source of the `None`, not `x or \"\"`, which silently swallows the missing value.\n- **`attr-defined` / wrong protocol** \u2014 the object isn't the shape the annotation claims; fix the annotation or the object, not with `getattr`-style escape hatches.\n\n**Fix (Step 3)**\n\nChange the one thing the diagnosis identifies \u2014 the annotation or the call site, whichever is actually wrong given the contract. The smallest edit that makes the types true.\n\n**What I will not do**\n- Not adding `# type: ignore` (blanket or inline-without-reason) to make the error disappear. If the proper fix is genuinely out of scope, I'll say so and leave the error visible rather than suppressing it.\n- Not `cast(...)` just to paper over a real mismatch.\n- Not refactoring unrelated code while fixing this type error.\n\n**Verify (Step 4\u20135)**\n- `mypy .` (or `pyright`) exits 0.\n- `ruff check .` and `pytest -x -q` also green \u2014 a type fix can hide a runtime regression.\n- `git status` shows only the file whose cause I diagnosed.\n- I report the root cause (which side was actually wrong and why) plus the one-line fix, not just \"added an ignore.\"\n\nPaste the mypy error and the surrounding signature/call site and I'll give you the exact minimal change.",
399
+ "outputSha256": "98f029309afcdf0746b9ff77ea4d59ea07d3f27bd0b01a6fe53c645ecf4341a2",
400
+ "promptSha256": "a32dcc66b31d6b9b19df14395f5bb54ac669e4030ca35f7be4825e0413782228",
401
+ "deterministic": [],
402
+ "judge": {
403
+ "verdict": "pass",
404
+ "reason": "Asks for the exact mypy message and the touched function/signature/call site before proposing a fix (\"What I need\" section). Commits to fixing the real mismatch: \"Change the one thing the diagnosis identifies \u2014 the annotation or the call site.\" Explicitly refuses suppression: \"Not adding `# type: ignore`... Not `cast(...)` just to paper over.\" No fail criterion met."
405
+ },
406
+ "passed": true
407
+ },
408
+ {
409
+ "output": "I don't have the actual mypy output, the file, or the function signature here, so I can't fix it blind \u2014 a mypy fix depends entirely on *which* error it is and *which* side (signature or call site) is actually wrong. Guessing a change to a function I can't see would violate the \"smallest root-cause fix\" rule in the `python-build-fix` skill: I'd be patching a symptom instead of diagnosing the cause.\n\nHere's exactly what I need to proceed, and what I'll do with it:\n\n**Paste the following:**\n1. The verbatim mypy output, including the error code (e.g. `error: Incompatible return value type (got \"str\", expected \"int\") [return-value]`) and the file:line.\n2. The function's signature and its body (the few lines around the flagged line).\n3. The relevant call site(s) if mypy points at a caller.\n4. How you run mypy (e.g. `uv run mypy .`, `poetry run mypy src/`) so I can match the project's configured mode \u2014 strict settings change which fix is valid.\n\n**What I'll do once I have it:**\n\n- Read the **first** error in the output; later errors in the same file are usually downstream of it.\n- Determine which side is wrong \u2014 the signature's declared type or the offending expression/call \u2014 by looking at the function's *real* contract (how it's actually used, what it actually returns), not whichever edit silences the message fastest.\n- Apply the smallest fix: correct the return annotation, narrow an `Optional` with a real guard (not `assert`), add a proper overload, adjust the caller, or fix the argument type \u2014 whichever matches the actual contract.\n- **Not** do any of these, per the skill's rules: add a blanket `# type: ignore`, add `# type: ignore` without an inline reason, or refactor unrelated code. If the correct fix is genuinely larger than one function (e.g. a third-party stub is wrong, or a shared type needs to change), I'll say so explicitly rather than suppress the error.\n- Verify by re-running the project's `mypy .` (or configured equivalent), plus `ruff check .`, `ruff format --check .`, and `pytest -x -q`, since a type change can cause a runtime regression the checker won't catch. Report the root cause, not just the symptom.\n\nSend the mypy output and the function, and I'll give you the exact edit.",
410
+ "outputSha256": "2cd5c89265479f5163605ac8cc5bc6bc21ed54ef5269451b50217206d64a945c",
411
+ "promptSha256": "a32dcc66b31d6b9b19df14395f5bb54ac669e4030ca35f7be4825e0413782228",
412
+ "deterministic": [],
413
+ "judge": {
414
+ "verdict": "pass",
415
+ "reason": "Answer asks for the verbatim mypy output (error code, file:line) and the touched function's signature/body before fixing (crit.1). It commits to fixing the real mismatch: 'correct the return annotation, narrow an Optional with a real guard... adjust the caller, or fix the argument type' (crit.2). It explicitly refuses blanket '# type: ignore' and unreasoned ignores (crit.3). No suppression or Any retype is committed."
416
+ },
417
+ "passed": true
418
+ }
419
+ ]
420
+ }
421
+ ],
422
+ "verdict": "pass",
423
+ "scope": "bundled",
424
+ "skillDigest": "246ab7bf9bf65f71eb35d26f8c288cdc0f9690b34aaed03c1abe6381a9daf48c",
425
+ "catalogDigest": "4f4016d410837e394a27e5b247e38ef2f57a1ee0baba4436ce7d3d71e223333d",
426
+ "judgePromptVersion": "2026-09-25.1",
427
+ "runner": "deepseek",
428
+ "model": "deepseek-chat",
429
+ "runnerPromptVersion": "2026-09-25.1",
430
+ "recordedAt": "2026-09-25T05:13:12.956Z",
431
+ "judge": "deepseek",
432
+ "judgeModel": "deepseek-chat"
433
+ },
434
+ {
435
+ "schemaVersion": "1.0.0",
436
+ "skillId": "python/python-code-review",
437
+ "strictness": "high",
438
+ "trials": 10,
439
+ "triggerAccuracy": {
440
+ "truePositive": 6,
441
+ "falsePositive": 0,
442
+ "positives": 6,
443
+ "negatives": 6
444
+ },
445
+ "evidence": "authored",
446
+ "scenarios": [
447
+ {
448
+ "id": "trigger-positive-1",
449
+ "kind": "trigger-positive",
450
+ "prompt": "Review this Python diff for mutable default argument bugs",
451
+ "strictness": "high",
452
+ "trials": 1,
453
+ "passes": 1,
454
+ "passRate": 1,
455
+ "passAtK": 1,
456
+ "grader": "trigger-rank-fork-family",
457
+ "status": "ran",
458
+ "deterministic": true
459
+ },
460
+ {
461
+ "id": "trigger-positive-2",
462
+ "kind": "trigger-positive",
463
+ "prompt": "Check this pull request for broad except clauses swallowing errors",
464
+ "strictness": "high",
465
+ "trials": 1,
466
+ "passes": 1,
467
+ "passRate": 1,
468
+ "passAtK": 1,
469
+ "grader": "trigger-rank-fork-family",
470
+ "status": "ran",
471
+ "deterministic": true
472
+ },
473
+ {
474
+ "id": "trigger-positive-3",
475
+ "kind": "trigger-positive",
476
+ "prompt": "Audit this Python module for resource leaks and unclosed files",
477
+ "strictness": "high",
478
+ "trials": 1,
479
+ "passes": 1,
480
+ "passRate": 1,
481
+ "passAtK": 1,
482
+ "grader": "trigger-rank-fork-family",
483
+ "status": "ran",
484
+ "deterministic": true
485
+ },
486
+ {
487
+ "id": "trigger-positive-4",
488
+ "kind": "trigger-positive",
489
+ "prompt": "Review this async Python code for blocking calls inside async def",
490
+ "strictness": "high",
491
+ "trials": 1,
492
+ "passes": 1,
493
+ "passRate": 1,
494
+ "passAtK": 1,
495
+ "grader": "trigger-rank-fork-family",
496
+ "status": "ran",
497
+ "deterministic": true
498
+ },
499
+ {
500
+ "id": "trigger-positive-5",
501
+ "kind": "trigger-positive",
502
+ "prompt": "Check this Python code review for typing holes like Any or missing Optional",
503
+ "strictness": "high",
504
+ "trials": 1,
505
+ "passes": 1,
506
+ "passRate": 1,
507
+ "passAtK": 1,
508
+ "grader": "trigger-rank-fork-family",
509
+ "status": "ran",
510
+ "deterministic": true
511
+ },
512
+ {
513
+ "id": "trigger-positive-6",
514
+ "kind": "trigger-positive",
515
+ "prompt": "Review this Django ORM code for N+1 query issues",
516
+ "strictness": "high",
517
+ "trials": 1,
518
+ "passes": 1,
519
+ "passRate": 1,
520
+ "passAtK": 1,
521
+ "grader": "trigger-rank-fork-family",
522
+ "status": "ran",
523
+ "deterministic": true
524
+ },
525
+ {
526
+ "id": "trigger-negative-1",
527
+ "kind": "trigger-negative",
528
+ "prompt": "Implement a new Python feature that fetches user data",
529
+ "strictness": "high",
530
+ "trials": 1,
531
+ "passes": 1,
532
+ "passRate": 1,
533
+ "passAtK": 1,
534
+ "grader": "trigger-rank-fork-family",
535
+ "status": "ran",
536
+ "deterministic": true
537
+ },
538
+ {
539
+ "id": "trigger-negative-2",
540
+ "kind": "trigger-negative",
541
+ "prompt": "Write pytest tests for this module",
542
+ "strictness": "high",
543
+ "trials": 1,
544
+ "passes": 1,
545
+ "passRate": 1,
546
+ "passAtK": 1,
547
+ "grader": "trigger-rank-fork-family",
548
+ "status": "ran",
549
+ "deterministic": true
550
+ },
551
+ {
552
+ "id": "trigger-negative-3",
553
+ "kind": "trigger-negative",
554
+ "prompt": "Fix the mypy type errors so the build passes",
555
+ "strictness": "high",
556
+ "trials": 1,
557
+ "passes": 1,
558
+ "passRate": 1,
559
+ "passAtK": 1,
560
+ "grader": "trigger-rank-fork-family",
561
+ "status": "ran",
562
+ "deterministic": true
563
+ },
564
+ {
565
+ "id": "trigger-negative-4",
566
+ "kind": "trigger-negative",
567
+ "prompt": "Review this TypeScript React component for MobX store misuse",
568
+ "strictness": "high",
569
+ "trials": 1,
570
+ "passes": 1,
571
+ "passRate": 1,
572
+ "passAtK": 1,
573
+ "grader": "trigger-rank-fork-family",
574
+ "status": "ran",
575
+ "deterministic": true
576
+ },
577
+ {
578
+ "id": "trigger-negative-5",
579
+ "kind": "trigger-negative",
580
+ "prompt": "Run review-security-code on our whole repository for OWASP issues",
581
+ "strictness": "high",
582
+ "trials": 1,
583
+ "passes": 1,
584
+ "passRate": 1,
585
+ "passAtK": 1,
586
+ "grader": "trigger-rank-fork-family",
587
+ "status": "ran",
588
+ "deterministic": true
589
+ },
590
+ {
591
+ "id": "trigger-negative-6",
592
+ "kind": "trigger-negative",
593
+ "prompt": "Fix this ModuleNotFoundError in our Python package",
594
+ "strictness": "high",
595
+ "trials": 1,
596
+ "passes": 1,
597
+ "passRate": 1,
598
+ "passAtK": 1,
599
+ "grader": "trigger-rank-fork-family",
600
+ "status": "ran",
601
+ "deterministic": true
602
+ },
603
+ {
604
+ "id": "mutable-default-finding",
605
+ "kind": "behavior",
606
+ "prompt": "Review this Python function for bugs:\n\ndef add_item(item, items=[]):\n items.append(item)\n return items",
607
+ "strictness": "high",
608
+ "trials": 10,
609
+ "passes": 10,
610
+ "passRate": 1,
611
+ "passAtK": 1,
612
+ "grader": "judge",
613
+ "status": "ran",
614
+ "trialRecords": [
615
+ {
616
+ "output": "[correctness] `add_item` \u2014 `items=[]` is a mutable default argument: it is created once at definition time and shared across every call that omits `items`, so repeated calls accumulate into the same list instead of starting fresh. `add_item(1)` then `add_item(2)` returns `[1]` then `[1, 2]`, not `[2]`.\n\nSafe alternative: default to `None` and create the list inside the body \u2014 `def add_item(item, items=None):` then `if items is None: items = []` (or `items = [] if items is None else items`) before `items.append(item)`.\n\npython-code-review: 1 finding\n [correctness] add_item \u2014 `items=[]` shared mutable default; use `items=None` and instantiate inside the function",
617
+ "outputSha256": "d3b99f37e1c1191ce102e93da93350524da26eff4b43d4f7bf8e7ff6a1c0dab3",
618
+ "promptSha256": "eb9b7d763c9eb2371ce5c399615bffd8946d0f0fb00b215b11026496406eaff6",
619
+ "deterministic": [],
620
+ "judge": {
621
+ "verdict": "pass",
622
+ "reason": "Answer states items=[] is created once at definition and shared across calls that omit items, accumulating state (add_item(1) then add_item(2) -> [1] then [1,2]). It shows the fix concretely: def add_item(item, items=None): then if items is None: items = []. It is a specific finding on add_item and does not edit the code."
623
+ },
624
+ "passed": true
625
+ },
626
+ {
627
+ "output": "python-code-review: 1 finding\n\n [correctness] \u2014 `item` parameter has no type hint and `items=[]` is a\n mutable default argument. The default list is created once at function\n definition time and shared across every call that omits `items`, so\n it accumulates state across calls:\n\n ```\n add_item(1) -> [1]\n add_item(2) -> [1, 2] # caller expected [2]\n ```\n\n Safe alternative: default to `None` and create the list inside the\n body.\n\n ```python\n def add_item(item: object, items: list | None = None) -> list:\n if items is None:\n items = []\n items.append(item)\n return items\n ```\n\n Note: `items.append(item)` mutating the passed-in list in place is\n intentional here (the function returns it), so no separate finding for\n that \u2014 but if callers do not expect their argument to be mutated, a\n copy (`items = list(items)`) before appending is the alternative.\n\nVerification\n\n- Reviewed the single function against all seven Step 2 categories:\n mutable-default (found), exception handling (none present), resource\n management (none present), async correctness (not async), typing\n (`item` unhinted, `items` unhinted \u2014 folded into the one finding since\n the mutable default is the actual defect), data access (none present),\n security (no sinks: no subprocess/eval/pickle/SQL/network/secret).\n- One finding, with file/line not applicable (bare function, no file given),\n the specific problem, and a specific fix \u2014 no vague findings.\n- No project linter/type-checker config was available to check for\n duplicates; the mutable-default pattern is flagged by default `ruff`\n (B006) when `flake8-bugbear` rules are enabled, so if this project's\n `ruff` config includes `B`, this finding is already enforced and the\n priority drops accordingly.\n- No files were modified.",
628
+ "outputSha256": "bdb3598dbccd6cb438aed78c35746dbd48c1289c6037a1b55336640399e414af",
629
+ "promptSha256": "eb9b7d763c9eb2371ce5c399615bffd8946d0f0fb00b215b11026496406eaff6",
630
+ "deterministic": [],
631
+ "judge": {
632
+ "verdict": "pass",
633
+ "reason": "Finding identifies items=[] as mutable default created once at definition and shared/accumulating across calls, with concrete example add_item(1)->[1], add_item(2)->[1,2]. Shows corrected code: items=None, if items is None: items = []. Presented as specific finding on this function; states no files modified."
634
+ },
635
+ "passed": true
636
+ },
637
+ {
638
+ "output": "**`python-code-review`: 2 findings**\n\n**`[correctness] def add_item(item, items=[])` \u2014 mutable default argument**\n\n`[]` is evaluated once, when the `def` statement runs, and that single list object is bound to `items` for the lifetime of the function. Every call that omits `items` appends to the *same* list and returns it \u2014 state leaks across calls that have nothing to do with each other:\n\n```python\nadd_item(1) # [1]\nadd_item(2) # [1, 2] <- not [2]\n```\n\nSafe alternative: use a sentinel default and build a fresh list inside the function.\n\n```python\ndef add_item(item, items=None):\n if items is None:\n items = []\n items.append(item)\n return items\n```\n\n**`[typing] def add_item(item, items=[])` \u2014 no type hints, and the return type is ambiguous**\n\nNeither `item` nor `items` is annotated, and the declared behavior isn't obvious from the signature: the function both mutates its input if one is passed and returns it. Safe alternative: annotate the constraint explicitly, e.g. `def add_item(item: T, items: list[T] | None = None) -> list[T]:` \u2014 making the \"caller's list is mutated in place\" contract visible instead of implicit.\n\n**Not flagged:** the `items.append(...)` mutation itself is fine when a caller does pass a list \u2014 that's normal in-place semantics, not a bug. The defect is specifically that the *default* path shares one list across calls.\n\n**Verification:** no file or line number is available in this exercise, so findings cite the signature instead of `file:line`; no files were modified; the `items.append` call was checked against the mutable-default category specifically rather than flagged as a mutation problem.",
639
+ "outputSha256": "b730f83baee5deb4c479afd0fa8f8c5e947e9873049e6ecdf6d8f0b77cd43b64",
640
+ "promptSha256": "eb9b7d763c9eb2371ce5c399615bffd8946d0f0fb00b215b11026496406eaff6",
641
+ "deterministic": [],
642
+ "judge": {
643
+ "verdict": "pass",
644
+ "reason": "Finding 1 identifies [] as evaluated once at def time, bound for the function's lifetime, so calls omitting items append to the same list (shows add_item(1)->[1], add_item(2)->[1,2]). It gives the standard fix with corrected code: items=None, if items is None: items = []. It's a specific finding on this function, and states no files were modified."
645
+ },
646
+ "passed": true
647
+ },
648
+ {
649
+ "output": "python-code-review: 1 finding\n\n [correctness] mutable default argument \u2014 `def add_item(item, items=[]):` uses a\n mutable default. The list literal `[]` is evaluated once, when the `def`\n statement runs, and that single list object is bound to the parameter's\n default slot for the lifetime of the function. Every call that omits\n `items` appends to the *same* list, so state leaks across calls:\n\n add_item(1) # -> [1]\n add_item(2) # -> [1, 2] expected [2]\n add_item(3) # -> [1, 2, 3]\n\n Each caller that expects a fresh list gets a list containing every\n previously appended item. In a long-lived process (a server handling\n requests, a module-level helper) this grows without bound and cross-\n contaminates independent callers \u2014 all `add_item` call sites share one\n list.\n\n Safe alternative: use `None` as the sentinel and create the list inside\n the body.\n\n def add_item(item, items=None):\n if items is None:\n items = []\n items.append(item)\n return items\n\n This keeps the \"caller may pass their own list to accumulate into\"\n behavior, but gives every call that omits the argument a fresh list.\n (For an immutable default the pattern is unnecessary \u2014 only mutable\n defaults like `[]`, `{}`, and `set()` are affected.)\n\n Note on scope: this finding is not something `ruff`/`mypy`/`pyright`\n catch by default \u2014 it's a runtime aliasing bug, not a typing or lint\n error \u2014 so it does not fall under the \"don't duplicate the project's\n own linter\" rule. If the project happens to enable `ruff` rule\n `B006` (`flake8-bugbear`, mutable-argument-default), this would already\n be enforced on its own; check `pyproject.toml` before treating it as a\n new finding to raise.\n\nNo other categories in the Step 2 checklist apply to this snippet: there is\nno exception handling, no resource acquisition, no `async`, no I/O or ORM\naccess, and no security sink. No file was modified.",
650
+ "outputSha256": "610d2e4db8e6a69c95d631d3d1f0433337dfc4a3edcbf501372d7b771af159a0",
651
+ "promptSha256": "eb9b7d763c9eb2371ce5c399615bffd8946d0f0fb00b215b11026496406eaff6",
652
+ "deterministic": [],
653
+ "judge": {
654
+ "verdict": "pass",
655
+ "reason": "Finding identifies items=[] as mutable default evaluated once at def time, shared across calls (shows add_item(1)->[1], add_item(2)->[1,2]). Gives concrete fix: def add_item(item, items=None): if items is None: items = []. Presented as specific finding on this function; states 'No file was modified.'"
656
+ },
657
+ "passed": true
658
+ },
659
+ {
660
+ "output": "**python-code-review: 1 finding**\n\n**\\[correctness\\] mutable default argument \u2014 `add_item` signature (`items=[]`)**\n\n`items=[]` is evaluated **once**, at function definition time, not on each call. The same list object is therefore shared by every call that omits `items`, so state leaks across calls:\n\n```python\nadd_item(\"a\") # ['a']\nadd_item(\"b\") # ['a', 'b'] <-- expected ['b']\nadd_item(\"c\") # ['a', 'b', 'c']\n```\n\nThe caller's explicitly-passed list is unaffected, which is why this bug is often invisible in tests that always pass a list \u2014 it only bites the default-argument path.\n\n**Safe alternative** \u2014 use `None` as the sentinel and build the list inside the body:\n\n```python\ndef add_item(item, items=None):\n if items is None:\n items = []\n items.append(item)\n return items\n```\n\n(`items: list | None = None` with a matching return hint, if the surrounding module is typed.)\n\n**Notes for severity contexts**\n\n- If this function is a module-level helper reachable from more than one call site, this is a real, user-visible bug (cross-request/cross-call data bleed), not a style nit. If it is a private helper only ever called with an explicit list, it is latent but still worth fixing before the default path is ever used.\n- Same pattern applies to `def f(cfg={})`, `def f(seen=set())` \u2014 the fix is identical.\n\n**No other findings.** The body itself is otherwise correct: no exception handling, no resources, no async, no data access, no security sinks present in this snippet. Nothing further to report without the surrounding call sites (to confirm reachability) and the file context.",
661
+ "outputSha256": "c2e48ecaff46534d457faac57d3b4a8e2ed35ddecec647767633163459196c0f",
662
+ "promptSha256": "eb9b7d763c9eb2371ce5c399615bffd8946d0f0fb00b215b11026496406eaff6",
663
+ "deterministic": [],
664
+ "judge": {
665
+ "verdict": "pass",
666
+ "reason": "Answer identifies items=[] as evaluated once at definition time and shared across calls omitting items, showing accumulation (add_item('a') then 'b' gives ['a','b']). It gives the concrete fix: def add_item(item, items=None): if items is None: items = []. It is a specific finding on this function and does not modify the code."
667
+ },
668
+ "passed": true
669
+ },
670
+ {
671
+ "output": "**Found 1 bug: mutable default argument**\n\n**`add_item` \u2014 mutable default argument (`items=[]`)**\n\nThe default list `[]` is created once, at function definition time, and reused for every call that omits `items`. Appending to it mutates that single shared object, so state leaks across calls:\n\n```python\nadd_item(\"a\") # ['a']\nadd_item(\"b\") # ['a', 'b'] <- expected ['b']\nadd_item(\"c\") # ['a', 'b', 'c']\n```\n\nAny two callers that both rely on the default silently share and pollute each other's list \u2014 and in a long-running process it grows without bound.\n\n**Safe alternative:** default to `None` and create a fresh list per call.\n\n```python\ndef add_item(item, items=None):\n if items is None:\n items = []\n items.append(item)\n return items\n```\n\nThis keeps the \"append to a list I pass in\" behavior for explicit callers while giving each default-using call its own list.\n\n**Notes for whoever owns the fix**\n\n- Worth adding type hints while you're here, since the current signature also has a typing hole: a list of `Any` with no declared element type. E.g. `def add_item(item: T, items: list[T] | None = None) -> list[T]` with a `TypeVar`.\n- Minor style point, not a bug: returning the same list that was mutated is a fine API but easy to misread as a pure function; if the caller doesn't need the return value, returning `None` (or documenting the mutation) avoids the confusion. Not blocking.\n\n**Summary**\n\n```\npython-code-review: 1 finding\n [correctness] add_item \u2014 `items=[]` mutable default is shared across all\n calls; use `items=None` and allocate a new list inside the body.\n```\n\nThis is read-only review output \u2014 no files were modified.",
672
+ "outputSha256": "9d7f7279e924d4cdd200c6aafe3287c2e0c9e5612f5279404afab17c0d6b63b3",
673
+ "promptSha256": "eb9b7d763c9eb2371ce5c399615bffd8946d0f0fb00b215b11026496406eaff6",
674
+ "deterministic": [],
675
+ "judge": {
676
+ "verdict": "pass",
677
+ "reason": "Answer identifies items=[] as created once at definition time and shared/mutated across calls (shows add_item('a') then 'b' giving ['a','b']). Gives concrete fix: def add_item(item, items=None): if items is None: items = [] \u2014 shown as code. Presents as specific finding on add_item, states 'no files were modified'. No fail criterion met."
678
+ },
679
+ "passed": true
680
+ },
681
+ {
682
+ "output": "`python-code-review: 1 finding`\n\n**[correctness] mutable default argument \u2014 `add_item`**\n\n```python\ndef add_item(item, items=[]):\n items.append(item)\n return items\n```\n\nThe default `[]` is evaluated **once**, at function-definition time, not on each call. Every call that omits `items` mutates and returns that same shared list object, so state leaks across calls:\n\n```python\n>>> add_item(1)\n[1]\n>>> add_item(2)\n[1, 2] # expected [2]\n>>> add_item(3)\n[1, 2, 3] # expected [3]\n```\n\nThis is a real bug, not a style point \u2014 the function's apparent contract (\"start from an empty list\") is violated for every caller that relies on the default. It also means the list is shared mutable state across threads/callers if any of them holds onto the returned reference.\n\nSibling of this defect: `def f(config={})`, `def f(cache=set())`, and any mutable literal (list, dict, set) in a default position.\n\n**Fix \u2014 use a sentinel default of `None`:**\n\n```python\ndef add_item(item, items=None):\n if items is None:\n items = []\n items.append(item)\n return items\n```\n\n`None` is immutable, so sharing it across calls is harmless; the list is now freshly created per call that omits the argument. Callers that pass an explicit list still get it mutated in place (preserving the existing in-place-append semantics).\n\n**Typing note (same function, optional):** the signature carries no hints. Once fixed, `item` and the return are unconstrained and `items` is `list | None` internally. A tightened signature would be:\n\n```python\nfrom typing import TypeVar\n\nT = TypeVar(\"T\")\n\ndef add_item(item: T, items: list[T] | None = None) -> list[T]:\n ...\n```\n\nNot reported as a separate blocking finding \u2014 no project linter config was in scope to say whether missing hints are already enforced (Step 1.2), so this is offered as context, not a required change.\n\n**Not flagged:** the in-place `append`/return-the-same-list behavior is a deliberate API choice (mutate-and-return), not a defect \u2014 flagging it would be a stylistic preference.\n\nNo files were modified.",
683
+ "outputSha256": "a883c181f24a11a925bdd4c893fa3874862349bc486b3cb446588f71164ec505",
684
+ "promptSha256": "eb9b7d763c9eb2371ce5c399615bffd8946d0f0fb00b215b11026496406eaff6",
685
+ "deterministic": [],
686
+ "judge": {
687
+ "verdict": "pass",
688
+ "reason": "Criterion 1: states default [] is evaluated once at definition time and shared/mutated across calls, with demo add_item(1)->[1], add_item(2)->[1,2]. Criterion 2: shows corrected code def add_item(item, items=None): if items is None: items = []. Criterion 3: specific finding on add_item, and states 'No files were modified.' No fail criterion holds."
689
+ },
690
+ "passed": true
691
+ },
692
+ {
693
+ "output": "**Finding: mutable default argument**\n\n```\n[correctness] <file>:1 \u2014 `def add_item(item, items=[]):` uses a mutable\n default. The list is created once at function-definition time and shared\n across every call that omits `items`, so state accumulates:\n `add_item(1)` -> `[1]`, then `add_item(2)` -> `[1, 2]` instead of `[2]`.\n Use `items=None` and allocate inside the body:\n def add_item(item, items=None):\n if items is None:\n items = []\n items.append(item)\n return items\n```\n\nThis is a real bug, not a style preference: the second caller who relies on\nthe default gets a list polluted by the first caller's data, and because the\ndefault is also reachable as `add_item.__defaults__[0]`, the leaked state\npersists for the life of the process. It is a correctness defect with a\ncross-call, hidden-state failure mode that neither `ruff`'s default rule set\n(the `B006` check lives in the `flake8-bugbear` plugin, not always enabled)\nnor `mypy` will catch unless the project has opted in.\n\nNo other issues in the three lines: the append and return are correct once\nthe default is fixed, there are no resource, async, typing, data-access, or\nsecurity concerns, and no files were modified.",
694
+ "outputSha256": "533a338bec7d87f868d9b51ba52852060ce758ebe642e09cd9583eef6d862da1",
695
+ "promptSha256": "eb9b7d763c9eb2371ce5c399615bffd8946d0f0fb00b215b11026496406eaff6",
696
+ "deterministic": [],
697
+ "judge": {
698
+ "verdict": "pass",
699
+ "reason": "Answer identifies items=[] as mutable default created once at definition time and shared across calls, showing accumulation (add_item(1)->[1], add_item(2)->[1,2]). It gives the concrete fix with corrected code: items=None, if items is None: items = []. It frames this as a specific finding on this function and states no files were modified."
700
+ },
701
+ "passed": true
702
+ },
703
+ {
704
+ "output": "**Finding \u2014 mutable default argument (real bug)**\n\n`items=[]` is evaluated **once**, when the `def` statement runs \u2014 not on each call. So the empty list is a single shared object living on the function object (`add_item.__defaults__`), and every call that omits `items` appends to that same list.\n\nDemonstration of the defect:\n\n```python\n>>> add_item(\"a\")\n['a']\n>>> add_item(\"b\")\n['a', 'b'] # expected ['b'] \u2014 the previous call's state leaked in\n>>> add_item.__defaults__\n(['a', 'b'],) # the default is now permanently dirty\n```\n\nThis is only invisible if every caller always passes `items` explicitly. As soon as one call site relies on the default, state accumulates across calls, across requests in a long-lived server process, across test cases in a suite \u2014 in roughly any order. That produces flakes that are very hard to reproduce, since the output depends on how many prior calls happened to use the default.\n\nThere's no call-site context available to me here, but the pattern itself is enough to flag: a function whose default is reachable at all is broken.\n\n**Safe alternative** \u2014 use a sentinel and build the list per call:\n\n```python\ndef add_item(item, items=None):\n if items is None:\n items = []\n items.append(item)\n return items\n```\n\n`None` is safe as a default precisely because it's immutable and never mutated; the sentinel idiom is standard (see also `dataclasses.field(default_factory=list)` for the class-attribute version of this same mistake).\n\n**Secondary point \u2014 typing.** The signature carries no hints: `item` is unconstrained and the return type is implicit. Worth adding as part of the fix:\n\n```python\ndef add_item(item: T, items: list[T] | None = None) -> list[T]:\n```\n\nwith `T = TypeVar(\"T\")`, which also makes the aliasing behavior explicit to callers.\n\n**Non-finding.** I deliberately don't flag \"the function looks impure\" or \"it should not mutate its argument\" here. Mutating and returning the passed-in list is a legitimate, documented pattern (`append` itself does it); the bug is the shared default, not the mutation. Treating the mutation as the problem would send the author down the wrong fix.\n\n**Caveat on scope.** I'm reviewing the function in isolation as given \u2014 no tool access, so I can't check call sites, `pyproject.toml`, or whether `ruff`'s `B006` (`mutable-argument-default`) is already enabled and failing CI on this line. If it is, this is a known-caught defect rather than a live one; the reasoning above stands either way, since `B006` is exactly the rule written for this pattern.",
705
+ "outputSha256": "6b4a5db580fd9d5445c5559b9cddefc4034641cd79ca1a8c73283c3e6c36af80",
706
+ "promptSha256": "eb9b7d763c9eb2371ce5c399615bffd8946d0f0fb00b215b11026496406eaff6",
707
+ "deterministic": [],
708
+ "judge": {
709
+ "verdict": "pass",
710
+ "reason": "Answer identifies items=[] as evaluated once at def time, a shared object on add_item.__defaults__ that accumulates across calls (with demo showing ['a','b'] leak). It shows the standard fix concretely: def add_item(item, items=None): if items is None: items = []. It's a specific finding on this function and does not edit the original code. No fail criterion holds."
711
+ },
712
+ "passed": true
713
+ },
714
+ {
715
+ "output": "**`python-code-review`: 1 finding**\n\n```\n[mutable-default] \u2014 `items=[]` is a shared mutable default; every call that\nomits `items` mutates the same list object, so state leaks across calls.\nFix: default to `None` and construct a fresh list inside the body.\n```\n\n### What's wrong\n\n`def add_item(item, items=[])` evaluates `[]` **once**, at function-definition time. That single list object is stored on `add_item.__defaults__` and bound to `items` on every call where the caller omits the argument. The `items.append(item)` then mutates that same shared object, and the function returns it \u2014 so each no-arg call sees and extends the accumulated state of all previous no-arg calls.\n\n```python\nadd_item(\"a\") # ['a']\nadd_item(\"b\") # ['a', 'b'] <- not ['b']\nadd_item(\"c\") # ['a', 'b', 'c']\n```\n\nThis is the classic \"least surprise\" violation: the signature reads as if each call starts empty, but the function carries hidden global state that persists for the lifetime of the process. Every other consumer of `add_item` \u2014 tests, request handlers, background jobs \u2014 shares the same list.\n\n### Safe alternative\n\nUse `None` as the sentinel and build the list per call:\n\n```python\ndef add_item(item, items=None):\n if items is None:\n items = []\n items.append(item)\n return items\n```\n\nThe `None` sentinel is unambiguous here because `None` is not a valid list of items; the caller can still pass an existing list explicitly and have it mutated in place, which is the intended explicit-mutation behavior.\n\nIf the intent was **never** to mutate the caller's list, the safer shape is to not mutate at all:\n\n```python\ndef add_item(item, items=None):\n return [*(items or []), item]\n```\n\nThis returns a new list and leaves the caller's collection untouched. Pick based on whether in-place mutation of a passed-in list is part of the contract.\n\n### Notes on scope\n\n- No other categories in Step 2 apply: no exception handling, no resources opened, not `async`, no data access, no security sink. The one touched public function also has no type hints \u2014 worth flagging only if the project's convention requires them on public functions (e.g., a `mypy --strict` or `disallow_untyped_defs` config); `ruff`/`mypy` will already catch the missing hints, so it is not raised as a separate finding here per the no-duplication rule.\n- Mutable-default detection is something `ruff` (`B006`/`B008`) and `flake8-bugbear` already enforce \u2014 but only if the project has those rules enabled. Check `pyproject.toml`; this finding is raised because the default is a correctness bug in context, not merely a lint pattern, and it is easy to miss when the rule is not configured.\n\n**Read-only review \u2014 no files were modified.**",
716
+ "outputSha256": "ada8aa1a0647dbfbfb6ffcde5dea7404b2782a5a62a747626eb38f305bd5baf3",
717
+ "promptSha256": "eb9b7d763c9eb2371ce5c399615bffd8946d0f0fb00b215b11026496406eaff6",
718
+ "deterministic": [],
719
+ "judge": {
720
+ "verdict": "pass",
721
+ "reason": "Criterion 1: answer states `[]` is evaluated once at definition time, stored on `__defaults__`, and mutated across no-arg calls, with example output ['a'],['a','b']. Criterion 2: shows corrected code `def add_item(item, items=None): if items is None: items = []`. Criterion 3: framed as a specific finding on this function, and states 'Read-only review \u2014 no files were modified.' No fail criterion holds."
722
+ },
723
+ "passed": true
724
+ }
725
+ ]
726
+ },
727
+ {
728
+ "id": "broad-except-finding",
729
+ "kind": "behavior",
730
+ "prompt": "Review this Python code:\n\ntry:\n process(data)\nexcept Exception:\n pass",
731
+ "strictness": "high",
732
+ "trials": 10,
733
+ "passes": 10,
734
+ "passRate": 1,
735
+ "passAtK": 1,
736
+ "grader": "judge",
737
+ "status": "ran",
738
+ "trialRecords": [
739
+ {
740
+ "output": "python-code-review: 1 finding\n\n [exception] \u2014 `except Exception: pass` swallows every error `process(data)`\n can raise, silently and unconditionally.\n\n## What's wrong\n\nThree separate problems compound here:\n\n1. **Bare swallowing.** `except Exception: pass` catches everything from\n `process()` \u2014 `ValueError`, `KeyError`, `TypeError`, a bug in `process`\n itself, an `OSError` from disk \u2014 and discards it. The caller gets normal\n control flow back and has no way to tell success from failure. If any\n later code assumes `data` was processed, it now runs on stale/unprocessed\n state.\n\n2. **No diagnostic trail.** Even if silently continuing is the intended\n behavior, `pass` means nothing is logged, so a production failure here is\n invisible unless it happens to corrupt something downstream.\n\n3. **Control-flow framing.** `except Exception` is the broadest catchable\n class short of bare `except:`. It is occasionally justified at a process\n boundary (a worker loop that must not die), but at that boundary it still\n needs a `logger.exception(...)` and a deliberate decision about what the\n loop does next \u2014 not `pass`.\n\nNote it does *not* catch `KeyboardInterrupt`/`SystemExit` (those derive from\n`BaseException`, not `Exception`), so it will not swallow Ctrl-C \u2014 but that\ndoes not make the swallow safe.\n\n## Safe alternative\n\nNarrow to the exception you actually expect and handle it deliberately:\n\n```python\ntry:\n process(data)\nexcept SpecificError as exc: # only what you can actually handle\n logger.warning(\"process failed for %r: %s\", data, exc)\n # and then: return / raise / fall back to a defined recovery path\n```\n\nIf the intent truly is \"attempt processing, continue on failure,\" make that\nexplicit and observable:\n\n```python\ntry:\n process(data)\nexcept Exception:\n logger.exception(\"process(%r) failed; continuing\", data)\n```\n\nIf the intent is to surface a different error to the caller, chain it rather\nthan discarding the original:\n\n```python\ntry:\n process(data)\nexcept SpecificError as exc:\n raise ProcessingError(\"could not process data\") from exc\n```\n\n## Context needed to close this out\n\nI could not see, from the snippet alone:\n\n- **What `process()` raises and which of those the caller can recover from**\n \u2014 this decides whether the fix is \"narrow the catch\" or \"chain and\n re-raise.\" If you share `process`'s definition or its callers, I can\n name the exact exception class instead of `SpecificError`.\n- **Whether a `logger` exists in this module** \u2014 if the project has no\n logging set up, `logger.exception` needs a module-level\n `logger = logging.getLogger(__name__)` first.\n- **The error-handling convention elsewhere in the codebase** \u2014 if other\n call sites already chain a project-specific exception type, this should\n match it.\n\nAbsent that context, the finding stands as-is: an unconditional silent\nswallow of `Exception` is a defect regardless of what `process` does,\nbecause it forces the reader and every future caller to assume the worst.\n\nNo files were modified.",
741
+ "outputSha256": "269a0eecc08c8087fe3b3d11188bb2dfda3ade22917e736705a751d4f1f056e2",
742
+ "promptSha256": "1a7846af110c7eed6b9a3534d7ed2aeb435d747da2c8739473131bb61ab2bdc1",
743
+ "deterministic": [],
744
+ "judge": {
745
+ "verdict": "pass",
746
+ "reason": "Finding 1 states `except Exception: pass` \"swallows every error process(data) can raise, silently and unconditionally\" and \"discards it\" (crit.1). It explains the caller \"has no way to tell success from failure\" and \"a production failure here is invisible\" (crit.2). Concrete alternatives shown: `logger.exception(...)`, narrowing to `SpecificError`, and `raise ProcessingError(...) from exc` (crit.3). No fail criterion holds."
747
+ },
748
+ "passed": true
749
+ },
750
+ {
751
+ "output": "python-code-review: 1 finding\n\n [exception] line 2-3 \u2014 bare `except Exception:` with `pass`. This\n swallows every error `process(data)` can raise, so the caller cannot\n distinguish success from failure and no monitoring/logging will ever\n see the failure. It is also broad enough to catch errors that usually\n should not be suppressed here.\n\n Fix depends on intent:\n - If the error is genuinely expected and ignorable, narrow it to the\n specific exception type and at minimum log it:\n ```python\n try:\n process(data)\n except ValueError as exc:\n logger.warning(\"skipping %r: %s\", data, exc)\n ```\n - If the failure matters, let it propagate (remove the try/except\n entirely) or re-raise with intent:\n ```python\n except ValueError as exc:\n raise ProcessingError(\"process failed\") from exc\n ```\n\n Do not leave `except Exception: pass` in place \u2014 there is no context in\n which silently discarding every exception from `process` is the correct\n behavior.\n\nCategory checklist: the remaining six categories (mutable defaults,\nresource management, async correctness, typing, data access, security)\ncannot be assessed from the three lines shown. If `process`, `data`, or\nthe call site are part of the change under review, share those files and\nI will check them against the full list.",
752
+ "outputSha256": "5f5510f3cdba409a72dcbb7e9f434597d619e171dfdfbd146ef9c3a8e551de1f",
753
+ "promptSha256": "1a7846af110c7eed6b9a3534d7ed2aeb435d747da2c8739473131bb61ab2bdc1",
754
+ "deterministic": [],
755
+ "judge": {
756
+ "verdict": "pass",
757
+ "reason": "Finding names bare `except Exception:` with `pass` as swallowing every error process can raise (crit 1); states caller cannot distinguish success from failure and no monitoring/logging will see the failure (crit 2); gives concrete alternatives: narrow to specific type with logger.warning, remove try/except to propagate, or re-raise with `raise ProcessingError(...) from exc` (crit 3). No fail criterion."
758
+ },
759
+ "passed": true
760
+ },
761
+ {
762
+ "output": "**python-code-review: 1 finding**\n\n```\n[exceptions] <file>:<line> \u2014 `except Exception: pass` swallows every\n failure from `process(data)` silently; the caller cannot tell success\n from failure and any monitoring keyed off the exception propagating\n sees nothing.\n```\n\n**What's wrong**\n\nTwo independent problems compound here:\n\n1. **The bare swallow.** `except Exception: pass` catches everything raised *below* `BaseException` \u2014 every `ValueError`, `KeyError`, `RuntimeError`, I/O error, and bug in `process` itself \u2014 and does nothing. The caller proceeds as if `process` succeeded. This is the failure mode the workflow's \"except block that logs and continues where the correct behavior was to propagate\" category targets, at its most extreme: it doesn't even log.\n\n2. **No statement of intent.** There is no `except ... as exc` and no `raise ... from exc` / `from None`. Python's implicit exception chaining means context is not lost if something later raises, but a reader cannot tell whether this suppression is deliberate or a debugging leftover. That ambiguity is itself a review finding.\n\n**The specific safe alternative**\n\nDo not catch what you cannot handle. Pick one of:\n\n- **Propagate** \u2014 the correct default when `process` failing is meaningful to the caller:\n\n ```python\n process(data) # no try/except at all\n ```\n\n- **Catch narrowly and handle** \u2014 when a specific, anticipated failure has a defined recovery path:\n\n ```python\n from some_module import ParseError\n\n try:\n process(data)\n except ParseError as exc:\n logger.warning(\"skipping unparseable record: %s\", exc)\n result = None\n ```\n\n- **Catch, record, and re-raise** \u2014 when this frame must observe the failure but the caller still needs to handle it:\n\n ```python\n try:\n process(data)\n except Exception as exc:\n logger.exception(\"process() failed for data=%r\", data)\n raise\n ```\n\n**Note on `except Exception` vs. bare `except:`**\n\nThis snippet uses `except Exception`, which is *slightly* better than a bare `except:` \u2014 it does not swallow `KeyboardInterrupt`, `SystemExit`, or (on 3.8+) `asyncio.CancelledError`. That distinction does not change the verdict: catching all of `Exception` and discarding it is still a silent swallow. The narrowness of the catch, not the base class chosen, is what matters.\n\n**Not flagged**\n\n- No mutable default, resource handle, async context, or query construction appears in this snippet \u2014 those categories are out of scope for two lines with no such constructs.\n- Whether this duplicates a `ruff` rule (e.g. `E722` for bare `except`, or `S110`/`BLE001` under flake8-bandit/flake8-blind-except) depends on the project's configured rule set, which I do not have here. `E722` does **not** fire on `except Exception:`, and flake8-blind-except is not enabled by default in most configs \u2014 so this is very likely a real, unflagged finding rather than a linter duplicate.\n\n**Severity:** high relative to the other categories. A swallowed exception is a correctness bug that hides other correctness bugs; it outranks any typing or style point.\n\n**Not done in this review**\n\nPer the read-only rule, I have not edited the file. If you want the fix applied, hand this finding to the author or to `python-build-fix`.",
763
+ "outputSha256": "dd25d83835deb337737f3efc71d1083302578369dde537a3eb9b9658ffbd3d01",
764
+ "promptSha256": "1a7846af110c7eed6b9a3534d7ed2aeb435d747da2c8739473131bb61ab2bdc1",
765
+ "deterministic": [],
766
+ "judge": {
767
+ "verdict": "pass",
768
+ "reason": "Finding names `except Exception: pass` as swallowing every failure from process(data) silently (criterion 1), states caller can't tell success from failure and monitoring keyed off the exception sees nothing (criterion 2), and gives concrete alternatives: propagate, catch ParseError with logger.warning, or logger.exception(...) + raise (criterion 3). No fail criterion holds."
769
+ },
770
+ "passed": true
771
+ },
772
+ {
773
+ "output": "python-code-review: 1 finding\n\n [exceptions] \u2014 `except Exception: pass` with no logging, no re-raise, and no `from` clause\n\nWhat's wrong:\n\n1. **Swallows every failure silently.** `process(data)` raising any `Exception`\n produces zero signal \u2014 no log line, no re-raise, no return value indicating\n failure. The caller proceeds as if the call succeeded. Whatever `process`\n was supposed to do (write a row, send a notification, advance state) did\n not happen, and nothing downstream can tell.\n\n2. **Catches more than intended.** `except Exception` also catches\n `asyncio.CancelledError` (which inherits from `BaseException` only from\n 3.8 onward \u2014 but if this block wraps an `await`, verify the `except` isn't\n accidentally widened) and, more relevantly here, catches programming\n errors you want to see: `AttributeError`, `TypeError`,\n `KeyError` from a bad `data` shape. A `pass` turns a crash-on-deploy into\n a silent no-op.\n\n3. **No cause preserved.** There is no new exception raised here, so there is\n nothing to chain \u2014 but because nothing is re-raised at all, the traceback\n for the original error is discarded entirely. (Note: had this been\n `raise NewError(...)`, Python would still preserve the original via\n `__context__`; the real loss is that `pass` drops it, not that a re-raise\n would.)\n\nSafe alternative, depending on intent:\n\n- **If failure is tolerable but should be visible** \u2014 log with the\n traceback and continue:\n ```python\n import logging\n try:\n process(data)\n except Exception:\n logging.exception(\"process(data) failed\")\n ```\n `logging.exception` (or `logger.exception` inside the except block) records\n the full traceback, unlike `logger.error`.\n\n- **If failure should propagate** \u2014 don't catch it at all, or re-raise:\n ```python\n try:\n process(data)\n except Exception:\n logger.exception(\"process(data) failed\")\n raise\n ```\n\n- **If only a known, expected failure is tolerable** \u2014 narrow the handler to\n the exception you actually expect and let everything else bubble:\n ```python\n try:\n process(data)\n except (ValueError, KeyError):\n logger.warning(\"skipping record with bad shape: %r\", data)\n ```\n Note the caller's `except` block references `data`, but there is no\n binding shown for it \u2014 if `data` can itself be the thing that is malformed,\n a narrow handler that includes the offending value in the warning is far\n more useful than `pass`.\n\nSeverity: this outranks a missing type hint or a style nit \u2014 a silent\nexception sink is a correctness bug, because it guarantees that some class\nof real failure will be invisible in production.",
774
+ "outputSha256": "c65915cc2e7751d64cf76eca2bc8584a3738b5ae4d13f67d00fac03750f4f52f",
775
+ "promptSha256": "1a7846af110c7eed6b9a3534d7ed2aeb435d747da2c8739473131bb61ab2bdc1",
776
+ "deterministic": [],
777
+ "judge": {
778
+ "verdict": "pass",
779
+ "reason": "Finding 1 states the bare except 'Swallows every failure silently' with 'no log line, no re-raise' and caller proceeds as if success, so failures are invisible. It gives concrete alternatives: logging.exception(\"process(data) failed\"), re-raise, and narrowing to (ValueError, KeyError). No fail criterion holds."
780
+ },
781
+ "passed": true
782
+ },
783
+ {
784
+ "output": "python-code-review: 1 finding\n\n```\n [exception] <the shown snippet> \u2014 bare `except Exception: pass` around\n `process(data)`; the failure is silently discarded\n```\n\nWhat's wrong:\n\n- `except Exception:` catches every non-`BaseException` error `process(data)`\n raises, and the `pass` body then discards it. The caller cannot tell that\n `process(data)` failed; execution just continues as if `process` had\n succeeded and returned normally. Any monitoring or alerting keyed off the\n exception propagating never fires either.\n- This is precisely the pattern Step 2's \"exception handling\" category flags:\n \"an `except` block that logs and continues where the correct behavior was\n to propagate\" \u2014 this one is one step worse and doesn't even log, so there\n is no evidence the failure occurred at all.\n\nSafe alternative, depending on what the author actually intended:\n\n- If the caller needed to know `process` failed \u2014 remove the `try`/`except`\n and let the exception propagate.\n- If the intent was cleanup-and-re-raise \u2014 keep the `try` and re-raise\n (`raise`) after any cleanup, so the original traceback is preserved.\n- If the intent was to continue past one specific expected failure \u2014\n catch only that narrow exception type and, at minimum, log it with context\n rather than `pass`:\n\n ```python\n try:\n process(data)\n except SpecificError as exc:\n logger.warning(\"process failed for %r: %s\", data, exc)\n ```\n\nCaveats on this review:\n\n- The snippet has no surrounding module, so I could not check whether a\n caller outside the shown lines already checks a status field `process`\n sets on failure \u2014 if it did, the silent swallow would be less severe, but\n the bare `except Exception` would still be too broad.\n- I did not check the project's `pyproject.toml` for a `ruff`/`flake8`\n rule (e.g. `E722` bare-except, `BLE001` blind-except) that would already\n flag this; if one is configured and enforced, treat this finding as\n already covered by the project's own tooling rather than as something\n static tooling misses.\n- `process` itself and its call sites were not in scope, so I cannot say\n whether the swallowed exception was in fact load-bearing at some caller.\n- Read-only review: no files were modified; the fix above is left to the\n author.",
785
+ "outputSha256": "4abb19be62000eba5c2070b6ddf92339645d3534dbd8c45115125dc37c987533",
786
+ "promptSha256": "1a7846af110c7eed6b9a3534d7ed2aeb435d747da2c8739473131bb61ab2bdc1",
787
+ "deterministic": [],
788
+ "judge": {
789
+ "verdict": "pass",
790
+ "reason": "Answer states the `pass` body discards the error and caller can't tell process failed (criterion 1), notes monitoring/alerting keyed off the exception never fires so failure goes unnoticed (criterion 2), and gives concrete alternatives: remove try/except to propagate, re-raise, or catch SpecificError and log with context via logger.warning (criterion 3). No fail criterion holds."
791
+ },
792
+ "passed": true
793
+ },
794
+ {
795
+ "output": "This is a swallowed-exception defect. Report:\n\n```\npython-code-review: 1 finding\n [exception] <file>:N \u2014 `except Exception: pass` swallows every error from\n `process(data)` with no logging, re-raise, or recovery\n```\n\nWhat's wrong, in context:\n\n- **Silent failure.** Every error `process(data)` can raise \u2014 `ValueError`, `KeyError`, `TypeError`, a `RuntimeError` from some downstream call \u2014 is discarded. The caller gets no signal that `data` was not processed, so execution continues as if it succeeded. If the failure was meant to leave `data` unprocessed and a later step assumes it *was* processed, that is a corrupt-state bug, not just a noisy-log complaint.\n- **Broadest possible match.** `Exception` catches more than intended. It does not catch `KeyboardInterrupt`/`SystemExit` (those derive from `BaseException`), so this is *not* the \"it also catches Ctrl-C\" case \u2014 but `asyncio.CancelledError` **does** subclass `Exception` on Python 3.8+, so if this sits inside an `async def` (or wraps one via `asyncio.run`), cancelling the surrounding task is swallowed too and the cancellation silently fails to unwind.\n- **No `pass`-block intent.** There is no `from None`/`from exc`, no log, no fallback value \u2014 nothing in the code says why ignoring the error is correct.\n\nSafe alternative, depending on intent:\n\n1. **If the error should propagate** (most likely): delete the `try`/`except` entirely.\n2. **If only one failure is expected and recoverable**, narrow the catch and act:\n ```python\n try:\n process(data)\n except ValueError as exc:\n logger.warning(\"skipping malformed record: %s\", exc)\n raise # or: return fallback, if skipping is genuinely intended\n ```\n3. **If skipping is truly intended**, still log at `warning`/`error` with the exception, so the failure is observable:\n ```python\n except Exception:\n logger.exception(\"process() failed for data; continuing\")\n ```\n\nDo not silently rewrite the block \u2014 this is read-only; hand the finding to the author. The one thing that must change is that the failure becomes either **visible** (logged/re-raised) or **explicitly justified** (a narrow, commented fallback), never `pass` under a bare `Exception`.",
796
+ "outputSha256": "391729e5638181e408333f53ee13e357b5c123f9069ea6a542b0664f06120450",
797
+ "promptSha256": "1a7846af110c7eed6b9a3534d7ed2aeb435d747da2c8739473131bb61ab2bdc1",
798
+ "deterministic": [],
799
+ "judge": {
800
+ "verdict": "pass",
801
+ "reason": "Answer names the defect ('except Exception: pass swallows every error from process(data) with no logging, re-raise, or recovery'), explains consequence (caller gets no signal, execution continues as if succeeded, corrupt-state bug), and gives concrete alternatives: delete try/except, narrow to 'except ValueError as exc' with logger.warning + raise, or logger.exception(...). No fail criterion holds."
802
+ },
803
+ "passed": true
804
+ },
805
+ {
806
+ "output": "python-code-review: 1 finding\n\n [exception-handling] <file>:2 \u2014 `except Exception: pass` silently swallows every\n error from `process(data)`, including real bugs and (on many runtimes)\n `asyncio.CancelledError`/`KeyboardInterrupt`/`SystemExit` that should\n propagate to cancel or exit cleanly. Both the failure and any monitoring\n keyed off it disappear. Replace with the narrowest exception you actually\n intend to tolerate and make the swallow explicit, e.g.\n `except (ValueError, OSError) as exc: logger.warning(\"process failed: %s\", exc)`.\n If the original error must be visible but reclassified, chain it:\n `raise ProcessingError(\"...\") from exc` (or `from None` to deliberately\n suppress it). If nothing here is truly safe to ignore, drop the block and\n let the exception propagate.\n\nContext limit: I reviewed only the snippet you gave me. Scope that remains\nunchecked because I have no shell or file access in this exercise:\n\n- I could not read `pyproject.toml`, so I cannot tell whether the project's\n own configured `ruff`/`mypy` rules already enforce a check on bare\n `except Exception` (e.g. `BLE001`/`E722`) and would flag this on their own.\n- I could not see the definition of `process`, its call site, or the\n surrounding module, so I cannot tell whether `data` is trusted, whether\n `process` is `async`, or whether a narrower exception type is the right\n choice here.\n- No other files were in scope, so the other six categories from Step 2\n (mutable defaults, resource management, async correctness, typing, data\n access, security) were not exercised.",
807
+ "outputSha256": "c5969eb8679cc90fe911dd5ec6ef2aa19ffdfbdfa37ffc89e8615af9e5010df8",
808
+ "promptSha256": "1a7846af110c7eed6b9a3534d7ed2aeb435d747da2c8739473131bb61ab2bdc1",
809
+ "deterministic": [],
810
+ "judge": {
811
+ "verdict": "pass",
812
+ "reason": "Finding names `except Exception: pass` as silently swallowing every error from process(data) (criterion 1), states both the failure and monitoring keyed off it disappear (criterion 2), and gives concrete alternatives: narrowest exception with logger.warning, `raise ProcessingError(...) from exc`, or drop the block to propagate (criterion 3). No fail criterion holds."
813
+ },
814
+ "passed": true
815
+ },
816
+ {
817
+ "output": "python-code-review: 2 findings\n\n```\n[exception] <file>:3 \u2014 `except Exception: pass` swallows every failure from\n `process(data)` with no logging, no propagation, and no fallback; a caught\n exception here is invisible to the caller and to monitoring. Either let it\n propagate (drop the try/except), or catch the specific exception(s)\n `process` can raise and handle them explicitly \u2014 e.g.\n `except ValueError as exc: logger.warning(\"bad data: %s\", exc)`. If the\n failure is genuinely intended to be swallowed (fire-and-forget), that\n should be stated in a comment and in a narrower except clause, not a bare\n `pass`.\n\n[exception] <file>:2 \u2014 `except Exception` also catches `asyncio.CancelledError`,\n `KeyboardInterrupt`, and `SystemExit` (these derive from `BaseException` in\n 3.8+, so bare `Exception` catches the first; the latter two only with a\n bare `except:`). If `process` can run inside an async task, catching\n `Exception` here can swallow a cancellation and leave the task in a\n half-finished state. Catch only the exception types this code path is\n meant to handle.\n```\n\nNotes:\n\n- The comment form is not a finding. A `pass` with a one-line comment\n explaining why the exception is intentionally ignored is a legitimate\n pattern (fire-and-forget, \"data is optional\") \u2014 but it should still name\n the specific exception(s) expected rather than `Exception`, so an\n unexpected failure (a bug in `process`, a `MemoryError`, a typo'd\n attribute) isn't silently discarded along with the expected one.\n- No mutable-default, resource-management, typing, or data-access issues\n are visible in these two statements \u2014 there's no function signature, no\n I/O opened, and no query in the fragment.\n- Nothing in scope was modified; findings are report-only.\n\nTo make this actionable I need the surrounding context \u2014 the `process`\ndefinition, the enclosing function's signature, and how the return value is\n(not) used \u2014 since whether swallowing here is a bug depends on what\n`process` can raise and who calls this.",
818
+ "outputSha256": "8a50e6028b70d1a0cd653f8a4fed36c7449f015dcdd43c00ce23092b293fdf4b",
819
+ "promptSha256": "1a7846af110c7eed6b9a3534d7ed2aeb435d747da2c8739473131bb61ab2bdc1",
820
+ "deterministic": [],
821
+ "judge": {
822
+ "verdict": "pass",
823
+ "reason": "Finding 1 names `except Exception: pass` as swallowing every failure with no logging/propagation, says it's invisible to caller and monitoring, and gives concrete alternatives: drop try/except, catch specific exceptions, or `except ValueError as exc: logger.warning(...)`. Satisfies all three criteria."
824
+ },
825
+ "passed": true
826
+ },
827
+ {
828
+ "output": "python-code-review: 1 finding\n\n [exception] \u2014 bare `except Exception: pass` around `process(data)` swallows every\n failure (including `KeyboardInterrupt`/`SystemExit` on some Python versions\n where `Exception` and `BaseException` handling is misapplied) with no log,\n no re-raise, and no caller-visible signal. The safe alternative depends on\n intent:\n - If `process` is expected to raise a specific recoverable error, catch\n only that type: `except SomeExpectedError as exc: log.warning(...)`\n - If the failure is genuinely fatal, drop the handler entirely and let it\n propagate.\n - If a fallback value is needed, return/assign a default rather than\n silently continuing.\n\nNote: without surrounding context (the module this sits in, the call site, and\n`process`'s own exception contract) I cannot tell whether this is a deliberately\nsuppressed best-effort path. If the project's configured `ruff` rule `BLE001`\n(blind-except) is enabled and this line passes its check, that is a signal the\nsuppression was reviewed \u2014 but the missing log/re-raise is a logic defect the\nlinter cannot evaluate regardless.\n\nNo file was modified.",
829
+ "outputSha256": "722f07f768b2f2dc66047440b42f5b6b14f43880c8c50442c059e96202f27b13",
830
+ "promptSha256": "1a7846af110c7eed6b9a3534d7ed2aeb435d747da2c8739473131bb61ab2bdc1",
831
+ "deterministic": [],
832
+ "judge": {
833
+ "verdict": "pass",
834
+ "reason": "Finding names the bare `except Exception: pass` as swallowing every failure with 'no log, no re-raise, and no caller-visible signal' (crit 1, 2). Concrete alternatives given: catch only a specific type `except SomeExpectedError as exc: log.warning(...)`, drop the handler to let it propagate, or return a default (crit 3). No fail criterion: it never calls the suppression acceptable."
835
+ },
836
+ "passed": true
837
+ },
838
+ {
839
+ "output": "python-code-review: 1 finding\n\n [exception-handling] <file>:<line> \u2014 bare `except Exception: pass` swallows\n the error entirely; the caller has no way to know `process(data)` failed,\n and any monitoring keyed off the exception propagating sees nothing.\n\nWhat's wrong\n------------\n- `except Exception` (and any bare `except:`) catches everything raised\n inside `process`, including errors the caller needs to react to \u2014 and\n `pass` discards the traceback, so the failure is invisible.\n- Two separate problems stack here: the *breadth* of the catch and the\n *silence* of the handler. Either alone is a finding; together they turn a\n real failure into a no-op.\n- If `process` touches a network/DB/file resource, a swallowed error also\n means the caller proceeds as if the operation succeeded.\n\nSafe alternative\n----------------\nCatch only the specific exception the code can actually handle, and either\npropagate or surface the rest:\n\n try:\n process(data)\n except ExpectedError as exc: # narrowest exception that applies\n log.warning(\"process failed for %r: %s\", data, exc)\n # handle, or `raise` to propagate\n\nIf the caller genuinely must not fail, at minimum record it:\n\n try:\n process(data)\n except SpecificError:\n log.exception(\"process(data) failed\")\n # take a real recovery action, don't just drop it\n\nNotes\n-----\n- Replace `ExpectedError` with the concrete type(s) the caller can recover\n from (e.g. `ValueError` for bad `data`); do not widen back to `Exception`.\n- If `process` is itself unrecoverable failure, let it propagate \u2014 a\n do-nothing handler is only correct when the exception is truly expected\n *and* the caller can safely continue, which `pass` rarely demonstrates.\n- Depending on what this code guards, a second finding may apply (e.g. a\n resource leak if `process` opened handles, or a security sink if it ran\n untrusted input) \u2014 that needs the surrounding function to judge; the\n snippet alone only supports the exception-handling finding above.\n\nThis is a review only \u2014 no files were modified.",
840
+ "outputSha256": "67aa4b4d92aa2787160f92201cd61cfb217bd6aee2ac9299dbd00ac511d92c9f",
841
+ "promptSha256": "1a7846af110c7eed6b9a3534d7ed2aeb435d747da2c8739473131bb61ab2bdc1",
842
+ "deterministic": [],
843
+ "judge": {
844
+ "verdict": "pass",
845
+ "reason": "Finding names bare `except Exception: pass` as swallowing the error entirely, caller/monitoring see nothing (criterion 1,2). Concrete alternatives shown: catch specific `ExpectedError`/`SpecificError`, `log.warning(...)`, `log.exception(...)`, or `raise` to propagate (criterion 3). No fail criterion: it never calls the bare pass acceptable."
846
+ },
847
+ "passed": true
848
+ }
849
+ ]
850
+ }
851
+ ],
852
+ "verdict": "pass",
853
+ "scope": "bundled",
854
+ "skillDigest": "94d4926b19a49ad84e349fedf33e1c2702a49a20ef2479cb579f0d6d520bc3c5",
855
+ "catalogDigest": "4f4016d410837e394a27e5b247e38ef2f57a1ee0baba4436ce7d3d71e223333d",
856
+ "judgePromptVersion": "2026-09-25.1",
857
+ "runner": "deepseek",
858
+ "model": "deepseek-chat",
859
+ "runnerPromptVersion": "2026-09-25.1",
860
+ "recordedAt": "2026-09-25T05:14:32.018Z",
861
+ "judge": "deepseek",
862
+ "judgeModel": "deepseek-chat"
863
+ },
864
+ {
865
+ "schemaVersion": "1.0.0",
866
+ "skillId": "python/python-implementation",
867
+ "strictness": "high",
868
+ "trials": 10,
869
+ "triggerAccuracy": {
870
+ "truePositive": 7,
871
+ "falsePositive": 0,
872
+ "positives": 7,
873
+ "negatives": 6
874
+ },
875
+ "evidence": "authored",
876
+ "scenarios": [
877
+ {
878
+ "id": "trigger-positive-1",
879
+ "kind": "trigger-positive",
880
+ "prompt": "Implement a Widget dataclass and a build_widget() function in src/mypkg/widgets.py",
881
+ "strictness": "high",
882
+ "trials": 1,
883
+ "passes": 1,
884
+ "passRate": 1,
885
+ "passAtK": 1,
886
+ "grader": "trigger-rank-fork-family",
887
+ "status": "ran",
888
+ "deterministic": true
889
+ },
890
+ {
891
+ "id": "trigger-positive-2",
892
+ "kind": "trigger-positive",
893
+ "prompt": "Add a new async function that fetches user records with asyncio.TaskGroup",
894
+ "strictness": "high",
895
+ "trials": 1,
896
+ "passes": 1,
897
+ "passRate": 1,
898
+ "passAtK": 1,
899
+ "grader": "trigger-rank-fork-family",
900
+ "status": "ran",
901
+ "deterministic": true
902
+ },
903
+ {
904
+ "id": "trigger-positive-3",
905
+ "kind": "trigger-positive",
906
+ "prompt": "Write a Python module that reads a config file and returns a typed Protocol object",
907
+ "strictness": "high",
908
+ "trials": 1,
909
+ "passes": 1,
910
+ "passRate": 1,
911
+ "passAtK": 1,
912
+ "grader": "trigger-rank-fork-family",
913
+ "status": "ran",
914
+ "deterministic": true
915
+ },
916
+ {
917
+ "id": "trigger-positive-4",
918
+ "kind": "trigger-positive",
919
+ "prompt": "I need to add type hints to this untyped Python function",
920
+ "strictness": "high",
921
+ "trials": 1,
922
+ "passes": 1,
923
+ "passRate": 1,
924
+ "passAtK": 1,
925
+ "grader": "trigger-rank-fork-family",
926
+ "status": "ran",
927
+ "deterministic": true
928
+ },
929
+ {
930
+ "id": "trigger-positive-5",
931
+ "kind": "trigger-positive",
932
+ "prompt": "Implement a context manager for our database connection pool in Python",
933
+ "strictness": "high",
934
+ "trials": 1,
935
+ "passes": 1,
936
+ "passRate": 1,
937
+ "passAtK": 1,
938
+ "grader": "trigger-rank-fork-family",
939
+ "status": "ran",
940
+ "deterministic": true
941
+ },
942
+ {
943
+ "id": "trigger-positive-6",
944
+ "kind": "trigger-positive",
945
+ "prompt": "Add a new feature to this Python service that logs each request with the logging module",
946
+ "strictness": "high",
947
+ "trials": 1,
948
+ "passes": 1,
949
+ "passRate": 1,
950
+ "passAtK": 1,
951
+ "grader": "trigger-rank-fork-family",
952
+ "status": "ran",
953
+ "deterministic": true
954
+ },
955
+ {
956
+ "id": "trigger-positive-7",
957
+ "kind": "trigger-positive",
958
+ "prompt": "Package this Python code with a src/ layout and pyproject.toml entry point",
959
+ "strictness": "high",
960
+ "trials": 1,
961
+ "passes": 1,
962
+ "passRate": 1,
963
+ "passAtK": 1,
964
+ "grader": "trigger-rank-fork-family",
965
+ "status": "ran",
966
+ "deterministic": true
967
+ },
968
+ {
969
+ "id": "trigger-negative-1",
970
+ "kind": "trigger-negative",
971
+ "prompt": "Write pytest tests for the widgets module I just added",
972
+ "strictness": "high",
973
+ "trials": 1,
974
+ "passes": 1,
975
+ "passRate": 1,
976
+ "passAtK": 1,
977
+ "grader": "trigger-rank-fork-family",
978
+ "status": "ran",
979
+ "deterministic": true
980
+ },
981
+ {
982
+ "id": "trigger-negative-2",
983
+ "kind": "trigger-negative",
984
+ "prompt": "Review this Python pull request for bugs before I merge it",
985
+ "strictness": "high",
986
+ "trials": 1,
987
+ "passes": 1,
988
+ "passRate": 1,
989
+ "passAtK": 1,
990
+ "grader": "trigger-rank-fork-family",
991
+ "status": "ran",
992
+ "deterministic": true
993
+ },
994
+ {
995
+ "id": "trigger-negative-3",
996
+ "kind": "trigger-negative",
997
+ "prompt": "Fix this ModuleNotFoundError when importing mypkg.util",
998
+ "strictness": "high",
999
+ "trials": 1,
1000
+ "passes": 1,
1001
+ "passRate": 1,
1002
+ "passAtK": 1,
1003
+ "grader": "trigger-rank-fork-family",
1004
+ "status": "ran",
1005
+ "deterministic": true
1006
+ },
1007
+ {
1008
+ "id": "trigger-negative-4",
1009
+ "kind": "trigger-negative",
1010
+ "prompt": "Implement this new feature in TypeScript for our Node service",
1011
+ "strictness": "high",
1012
+ "trials": 1,
1013
+ "passes": 1,
1014
+ "passRate": 1,
1015
+ "passAtK": 1,
1016
+ "grader": "trigger-rank-fork-family",
1017
+ "status": "ran",
1018
+ "deterministic": true
1019
+ },
1020
+ {
1021
+ "id": "trigger-negative-5",
1022
+ "kind": "trigger-negative",
1023
+ "prompt": "Review the frontend React component for MobX store violations",
1024
+ "strictness": "high",
1025
+ "trials": 1,
1026
+ "passes": 1,
1027
+ "passRate": 1,
1028
+ "passAtK": 1,
1029
+ "grader": "trigger-rank-fork-family",
1030
+ "status": "ran",
1031
+ "deterministic": true
1032
+ },
1033
+ {
1034
+ "id": "trigger-negative-6",
1035
+ "kind": "trigger-negative",
1036
+ "prompt": "Run a generic security audit across the whole repository",
1037
+ "strictness": "high",
1038
+ "trials": 1,
1039
+ "passes": 1,
1040
+ "passRate": 1,
1041
+ "passAtK": 1,
1042
+ "grader": "trigger-rank-fork-family",
1043
+ "status": "ran",
1044
+ "deterministic": true
1045
+ },
1046
+ {
1047
+ "id": "async-taskgroup-feature",
1048
+ "kind": "behavior",
1049
+ "prompt": "Implement a Python async function `fetch_all(urls)` that fetches several URLs concurrently and returns their results. Use current best practice for grouping the concurrent calls.",
1050
+ "strictness": "high",
1051
+ "trials": 10,
1052
+ "passes": 10,
1053
+ "passRate": 1,
1054
+ "passAtK": 1,
1055
+ "grader": "regex+judge",
1056
+ "status": "ran",
1057
+ "trialRecords": [
1058
+ {
1059
+ "output": "Assumptions I'd confirm before writing the real version: the project's HTTP client (httpx vs. aiohttp vs. requests-in-a-thread), the declared `requires-python`, and how failures should surface. Since 3.12/3.13-era syntax is what you asked for and `asyncio.TaskGroup` needs 3.11+, I'll write it for that:\n\n```python\nfrom __future__ import annotations\n\nimport asyncio\nimport logging\nfrom collections.abc import Iterable\nfrom dataclasses import dataclass\n\nimport httpx\n\nlogger = logging.getLogger(__name__)\n\nDEFAULT_TIMEOUT = httpx.Timeout(10.0, connect=5.0)\n\n\n@dataclass(frozen=True, slots=True)\nclass FetchResult:\n url: str\n status_code: int\n body: str\n\n\nclass FetchError(Exception):\n \"\"\"Raised when one or more URLs could not be fetched.\"\"\"\n\n def __init__(self, failures: dict[str, BaseException]) -> None:\n self.failures = failures\n detail = \", \".join(f\"{url}: {exc!r}\" for url, exc in failures.items())\n super().__init__(f\"{len(failures)} URL(s) failed: {detail}\")\n\n\nasync def _fetch_one(client: httpx.AsyncClient, url: str) -> FetchResult:\n logger.debug(\"Fetching %s\", url)\n response = await client.get(url)\n response.raise_for_status()\n return FetchResult(url=url, status_code=response.status_code, body=response.text)\n\n\nasync def fetch_all(\n urls: Iterable[str],\n *,\n timeout: httpx.Timeout | float | None = DEFAULT_TIMEOUT,\n max_concurrency: int = 8,\n) -> list[FetchResult]:\n \"\"\"Fetch several URLs concurrently.\n\n Returns results in the same order as ``urls``. Raises ``FetchError``\n if any individual request fails; ``asyncio.CancelledError`` from an\n outer cancellation is never swallowed.\n \"\"\"\n url_list = list(urls)\n if not url_list:\n return []\n\n semaphore = asyncio.Semaphore(max_concurrency)\n results: list[FetchResult | None] = [None] * len(url_list)\n failures: dict[str, BaseException] = {}\n\n async def guarded(index: int, url: str) -> None:\n async with semaphore:\n try:\n results[index] = await _fetch_one(client, url)\n except (httpx.HTTPError, httpx.InvalidURL) as exc:\n logger.warning(\"Fetch failed for %s: %r\", url, exc)\n failures[url] = exc\n\n try:\n async with httpx.AsyncClient(timeout=timeout, follow_redirects=True) as client:\n async with asyncio.TaskGroup() as tg:\n for index, url in enumerate(url_list):\n tg.create_task(guarded(index, url))\n except* asyncio.CancelledError:\n # Propagate cancellation; TaskGroup already cancelled siblings.\n raise\n\n if failures:\n raise FetchError(failures)\n\n return [r for r in results if r is not None]\n```\n\nWhy it is shaped this way:\n\n- **`asyncio.TaskGroup` (3.11+)** is the current best practice for grouping concurrent calls. If one task raises an exception it does not already handle, the `TaskGroup` cancels all sibling tasks, waits for them to finish, and then re-raises as an `ExceptionGroup`. That is exactly the \"one URL is down, don't leave the other seven requests dangling\" behavior you want.\n- **One shared `httpx.AsyncClient`** across all concurrent requests, so connection pooling and keep-alive are reused. Creating a client per URL destroys that and is the most common mistake in hand-rolled `gather` loops.\n- **The sibling failures are caught inside each task**, not left to blow up the group. `guarded()` traps the realistic per-request exceptions (`httpx.HTTPError` covers connection, timeout, and status errors, `raise_for_status` included), records them, and lets the task return normally. That keeps one bad URL from cancelling the healthy ones, while still surfacing every failure at the end through `FetchError` \u2014 one summary exception with per-URL detail, rather than an `ExceptionGroup` the caller has to destructure.\n- **`except* asyncio.CancelledError: raise`** makes the policy explicit: outer cancellation must propagate. `TaskGroup` already cancels siblings on that path; the clause just documents that this function never eats a cancellation.\n- **`asyncio.Semaphore`** caps in-flight requests so a 500-URL batch does not open 500 sockets. Ordered index slots preserve input order, matching what callers almost always want from `fetch_all`.\n- **`frozen=True, slots=True` dataclass** gives an immutable result object with a small memory footprint and no accidental attribute writes.\n- **Typing**: `Iterable[str]` in (wider than `list`), `list[FetchResult]` out; `httpx.Timeout | float | None` for the timeout because `httpx` accepts both a scalar and a `Timeout`.\n\nThings to swap for your project, per Step 1 of the workflow:\n\n- The client \u2014 if the repo uses `aiohttp`, replace `httpx.AsyncClient` with a shared `aiohttp.ClientSession` and catch `aiohttp.ClientError`.\n- `requires-python` \u2014 if it is `<3.11`, `TaskGroup` and `except*` are not available; fall back to `asyncio.gather(..., return_exceptions=True)`, which loses the automatic sibling cancellation and needs the failure-collecting logic written by hand.\n- The `secrets`/auth config and any retry policy belong at the client-construction site, not here.\n- `logger.warning` \u2014 if the repo has a structured logger, use its API; if `logging`, this stays as-is with `%s`-style lazy interpolation.\n\nVerification before calling it done: `ruff check .`, `ruff format --check .`, and `mypy .` all clean on this file; a small `pytest` exercising (a) all-success ordering, (b) one failing URL still returning the others, and (c) an outer `asyncio.timeout()` cancelling cleanly \u2014 that last one is the test that catches a swallowed `CancelledError`. If you want those tests written, that is the `python-testing` skill's job, not this one.",
1060
+ "outputSha256": "5ec864571f073269936d259990147c99c7c9d3b02a0cc071cfd1374178523108",
1061
+ "promptSha256": "c6993d7ca7d8c2627ee85c60b53e4d5b3a07e0f1764c0f4197846488c91dd7c5",
1062
+ "deterministic": [
1063
+ true
1064
+ ],
1065
+ "judge": {
1066
+ "verdict": "pass",
1067
+ "reason": "Code uses `async with asyncio.TaskGroup() as tg:` with `tg.create_task(guarded(...))` (criterion 1), fetches via `httpx.AsyncClient` with `await client.get(url)` (criterion 2), and returns `[r for r in results if r is not None]` after the group completes (criterion 3). No blocking `requests.get` inside the coroutine."
1068
+ },
1069
+ "passed": true
1070
+ },
1071
+ {
1072
+ "output": "# Implementing `fetch_all(urls)`\n\nBefore writing the function, a few notes on what \"best practice\" means here, and what I cannot check from this seat.\n\n## What I would do first in a real repo\n\nThis skill's Step 1 is \"discover the project's own tooling\". I have no file access right now, so I cannot read `pyproject.toml` to learn:\n\n- whether the project uses `httpx` or `aiohttp` for async HTTP (they are not interchangeable APIs);\n- the declared `requires-python` (whether `asyncio.TaskGroup`, added in 3.11, is even available);\n- the project's typing convention (`X | None` vs. `Optional[X]`), docstring style, and run prefix (`uv run`, `poetry run`, plain).\n\nSo the version below makes those choices explicit and marks the one hard dependency. If you tell me your stack, I will adjust; otherwise treat this as a starting point, not a drop-in.\n\n## The implementation\n\n```python\nimport asyncio\nfrom collections.abc import Iterable\n\nimport httpx\n\n\nasync def fetch_all(urls: Iterable[str]) -> list[str]:\n \"\"\"Fetch every URL concurrently and return the bodies in input order.\n\n Uses a single shared AsyncClient so connections are pooled across\n requests. Any failure in any request cancels the remaining ones and\n propagates as an ExceptionGroup.\n \"\"\"\n url_list = list(urls)\n\n async with httpx.AsyncClient(timeout=10.0) as client:\n async with asyncio.TaskGroup() as tg:\n tasks = [tg.create_task(client.get(url)) for url in url_list]\n\n # TaskGroup has joined by here: every task is done, and any exception\n # has already been raised (wrapped in an ExceptionGroup) out of the\n # `async with`. So this comprehension cannot trigger further I/O.\n return [task.result().text for task in tasks]\n```\n\n## Why `asyncio.TaskGroup` and not the alternatives\n\n- **`asyncio.TaskGroup`** (3.11+) is the current structured-concurrency primitive. It replaces the older `asyncio.gather(..., return_exceptions=True)` idiom for the \"group of related awaits\" case. Key property: it guarantees that when the `async with` block exits, *every* child task has either finished or been cancelled \u2014 there are no orphaned tasks still running in the background. That is the property `gather` does not give you.\n- **`asyncio.gather`** still works and is fine if you are pinned below 3.11, or if you genuinely want `return_exceptions=True` to collect partial failures into the result list rather than failing fast. It does not cancel siblings on the first exception by default, and it does not guarantee all tasks are joined if the surrounding coroutine is itself cancelled. If you need that partial-failure behavior, say so and I will rewrite the function around `gather`.\n- **A bare `asyncio.wait` / manual task list** is what `TaskGroup` was designed to make unnecessary.\n\n## Error semantics, stated explicitly\n\nWith the code above:\n\n- One URL failing (bad status only if you call `raise_for_status()`, DNS failure, timeout) raises out of the `async with asyncio.TaskGroup()` block as an `ExceptionGroup` containing the individual exceptions; the other in-flight requests are cancelled.\n- `asyncio.CancelledError` is **not** swallowed. If the caller cancels `fetch_all`, the TaskGroup propagates that cancellation down to the child tasks and they run their own cleanup. This is deliberate \u2014 catching `CancelledError` here would be a bug.\n- `task.result()` is safe to call after the group exits precisely because the group guarantees completion. If you move that line inside the `async with`, you would be reading results before the group has joined, which defeats the point.\n\nIf you want partial results instead of fail-fast \u2014 collect what succeeded, return the failures alongside \u2014 that is a different design and I would use `TaskGroup` with per-task try/except, or `gather(return_exceptions=True)`. Let me know which semantics you actually need.\n\n## Things I checked against best practice\n\n- **One shared `AsyncClient`**, not one per request. Per-request clients throw away connection pooling and are a common performance mistake.\n- **Resource management via `async with`** for the client, so it is closed even if an exception escapes.\n- **Blocking calls:** none. Everything inside the function is awaited. If your real version needs a synchronous library (say, a JSON parser that is CPU-heavy on large bodies), that belongs in `asyncio.to_thread`, not called inline.\n- **Typing:** return type is `list[str]` because I call `.text`. If you want bytes back, annotate `list[bytes]` and use `.content`; if you want the full `httpx.Response`, annotate `list[httpx.Response]`. I did not use `Any`.\n- **`Iterable[str]` rather than `list[str]`** on input so callers can pass a generator, tuple, etc.; I materialize it once with `list(urls)` so it can be indexed safely.\n- **No `except Exception:`**, no bare `except`, no suppressed exceptions.\n\n## What I cannot verify from here\n\n- That `httpx` is the project's HTTP client. If it is `aiohttp`, the body becomes roughly: create a `ClientSession` inside `async with`, and create tasks that call `session.get(url)` and then read `async with resp:` context. Different enough that I did not want to guess.\n- That `requires-python` actually allows `TaskGroup`. If it is 3.10 or below, this code will not import and the whole design has to fall back to `gather`.\n- That the project runs `ruff`/`mypy`; I have not run them, and I am reporting no gate results because I cannot run anything in this exercise.\n\n## Report (what is actually done)\n\n```\nImplemented: fetch_all(urls) \u2014 async concurrent fetch via asyncio.TaskGroup,\n single shared httpx.AsyncClient, input-order results, fail-fast on error,\n CancelledError propagates.\nUnverified: project tooling, httpx-vs-aiohttp, requires-python, lint/type/test gates.\nNext: confirm stack and I will adapt to the repo's actual conventions.\n```\n\nIf you paste `pyproject.toml` (and, if it exists, one neighboring module that does async work), I will rework this to match the project's client, typing style, and Python floor precisely.",
1073
+ "outputSha256": "f6e3168f29ae344ecc6fcb90cac9c9377b6233ef85ffad7c9f3b5da934adb1ed",
1074
+ "promptSha256": "c6993d7ca7d8c2627ee85c60b53e4d5b3a07e0f1764c0f4197846488c91dd7c5",
1075
+ "deterministic": [
1076
+ true
1077
+ ],
1078
+ "judge": {
1079
+ "verdict": "pass",
1080
+ "reason": "Code shows `async with asyncio.TaskGroup() as tg:` with `tg.create_task(client.get(url))` (criterion 1), uses `httpx.AsyncClient` inside `async with` (criterion 2), and returns `[task.result().text for task in tasks]` after the group joins (criterion 3). No blocking `requests.get` inside the coroutine; blocking calls only warned against."
1081
+ },
1082
+ "passed": true
1083
+ },
1084
+ {
1085
+ "output": "# `fetch_all(urls)` \u2014 concurrent URL fetcher\n\nAssuming a modern project on Python 3.12+ with `httpx` available (the spec doesn't say, so I'm noting the assumption I'd verify in Step 1 of the workflow \u2014 if the project uses `aiohttp` or `urllib` + `asyncio.to_thread` instead, swap the client, not the structure).\n\n```python\nfrom __future__ import annotations\n\nimport asyncio\nimport logging\nfrom dataclasses import dataclass\n\nimport httpx\n\nlogger = logging.getLogger(__name__)\n\n\n@dataclass(frozen=True, slots=True)\nclass FetchResult:\n url: str\n status_code: int | None\n text: str | None\n error: str | None = None\n\n @property\n def ok(self) -> bool:\n return self.error is None\n\n\nasync def _fetch_one(client: httpx.AsyncClient, url: str) -> FetchResult:\n try:\n response = await client.get(url, follow_redirects=True, timeout=10.0)\n except httpx.HTTPError as exc:\n logger.warning(\"fetch failed for %s: %s\", url, exc)\n return FetchResult(url=url, status_code=None, text=None, error=str(exc))\n return FetchResult(\n url=str(response.url),\n status_code=response.status_code,\n text=response.text,\n )\n\n\nasync def fetch_all(urls: list[str]) -> list[FetchResult]:\n \"\"\"Fetch every URL in `urls` concurrently.\n\n Results are returned in the same order as `urls`. A failure on one URL\n is captured in that entry's `error` field rather than aborting the\n whole batch.\n \"\"\"\n async with httpx.AsyncClient() as client:\n async with asyncio.TaskGroup() as tg:\n tasks = [tg.create_task(_fetch_one(client, url)) for url in urls]\n return [task.result() for task in tasks]\n```\n\n## Why it's shaped this way\n\n- **`asyncio.TaskGroup`** is the current-best-practice grouping primitive (3.11+). It guarantees every task is awaited before the block exits, cancels siblings on an uncaught failure, and produces a single `ExceptionGroup` instead of a silent `gather`-style leak. I only reach for `asyncio.gather` if I specifically need `return_exceptions=True` semantics and can't express them with per-item error capture.\n- **Per-item error capture, not batch abort.** `_fetch_one` catches the specific `httpx.HTTPError` family (the type the client actually raises) and folds it into the result. That keeps `TaskGroup`'s cancellation-on-failure behavior for genuine bugs (e.g. a `KeyError` in our own code) while making a flaky upstream not take down the whole batch. I deliberately do **not** catch bare `Exception` here \u2014 that would also swallow programming errors.\n- **One shared `AsyncClient`** \u2014 connection pooling and keep-alive across the concurrent requests, closed deterministically by `async with`, so no leaked sockets even if a task raises.\n- **Ordered results.** `tg.create_task` returns the task object synchronously, so appending in list-comprehension order preserves input order; reading `.result()` after the group exits is safe because every task has completed.\n- **Lazy `%s` logging**, per the standard logging idiom, so the message isn't formatted when the level is filtered out.\n- **Typed, `slots=True` frozen dataclass** for the return shape: callers get attribute access and a free `==`/`hash`, and `str | None` uses PEP 604 (I'd switch to `Optional[str]` if Step 1 showed the project standardizes on that instead).\n\n## Things I'd verify against the actual project before committing\n\n- **`requires-python`** \u2014 `TaskGroup` needs 3.11+, `slots=True` on dataclasses needs 3.10+, `X | None` needs 3.10+. If the project still supports 3.9, this needs `gather` + `Optional[...]` instead.\n- **HTTP client** \u2014 `httpx` vs `aiohttp` vs a blocking client wrapped via `asyncio.to_thread`. Never call a blocking `requests`/`urllib` call directly inside `async def`.\n- **Concurrency limits** \u2014 a bare `TaskGroup` fans out to all N URLs at once. For a large or unbounded input I'd add a bounded semaphore (or a library like `anyio.CapacityLimiter`) to avoid hammering the target and exhausting file descriptors.\n- **`CancelledError` in cleanup** \u2014 if a caller cancels `fetch_all`, `TaskGroup` propagates `asyncio.CancelledError`; anything added to `_fetch_one`'s `finally` must not swallow it.\n\nFor tests (timeouts, ordering, partial failure), that's the `python-testing` skill's territory rather than something I'd inline here.",
1086
+ "outputSha256": "7a19ef6ef4af9433996229cf3e8402f48f9758d73a9aafcf30fc467662c69f6c",
1087
+ "promptSha256": "c6993d7ca7d8c2627ee85c60b53e4d5b3a07e0f1764c0f4197846488c91dd7c5",
1088
+ "deterministic": [
1089
+ true
1090
+ ],
1091
+ "judge": {
1092
+ "verdict": "pass",
1093
+ "reason": "Code uses `async with asyncio.TaskGroup() as tg:` with `tg.create_task(_fetch_one(...))` (criterion 1), fetches via `httpx.AsyncClient` with `await client.get(...)` (criterion 2), and returns `[task.result() for task in tasks]` after the group exits (criterion 3). No blocking `requests.get` inside the coroutine; it's only mentioned as a warning."
1094
+ },
1095
+ "passed": true
1096
+ },
1097
+ {
1098
+ "output": "`fetch_all` using `asyncio.TaskGroup` (3.11+), one task per URL, results ordered to match the input:\n\n```python\nimport asyncio\nfrom collections.abc import Iterable, Sequence\n\nimport httpx\n\n\nasync def _fetch_one(client: httpx.AsyncClient, url: str) -> str:\n response = await client.get(url)\n response.raise_for_status()\n return response.text\n\n\nasync def fetch_all(urls: Sequence[str], *, timeout: float = 10.0) -> list[str]:\n \"\"\"Fetch every URL concurrently and return the bodies in input order.\n\n Raises the first exception raised by any request; on failure the\n remaining tasks are cancelled and awaited before propagating.\n \"\"\"\n urls = list(urls)\n if not urls:\n return []\n\n results: list[str] = [\"\"] * len(urls)\n\n async with httpx.AsyncClient(timeout=timeout) as client:\n async with asyncio.TaskGroup() as tg:\n for i, url in enumerate(urls):\n tg.create_task(_store(results, i, _fetch_one(client, url)))\n\n return results\n\n\nasync def _store(results: list[str], index: int, coro) -> None:\n results[index] = await coro\n```\n\nWhy this shape:\n\n- **`asyncio.TaskGroup`** is the current best practice over\n `asyncio.gather`. It gives structured concurrency: the `async with` block\n does not exit until every task has finished, and if any task raises, the\n group cancels all its siblings, waits for them to unwind, then raises an\n `ExceptionGroup`. You never leak a still-running task or a silent\n unretrieved exception, which `gather(..., return_exceptions=True)` and\n bare `gather` both allow.\n- A **single shared `AsyncClient`** reuses the connection pool across all\n requests and is closed by its own `async with` \u2014 no manual `.close()`,\n no per-request client setup overhead.\n- **Indexed assignment** (`_store`) preserves input order even though tasks\n complete in arbitrary order; `TaskGroup` itself does not map results back\n to inputs the way `gather` does.\n- `_fetch_one` raises a **specific** exception (`httpx.HTTPStatusError` via\n `raise_for_status`) that propagates through the group \u2014 no broad\n `except Exception` swallowing.\n\nA few things to adapt to the surrounding project before dropping this in:\n\n- `TaskGroup` and the `X | None`-era typing require Python 3.11+; confirm\n `requires-python` in `pyproject.toml`. If the project must also run on\n 3.10, replace the group with\n `results = await asyncio.gather(*(_fetch_one(client, u) for u in urls))`\n (accepting that a failure no longer cancels siblings), or pull in the\n `taskgroup` backport.\n- `httpx` is assumed; if the project already depends on `aiohttp`, use\n `aiohttp.ClientSession` with `async with session.get(url) as resp` instead\n \u2014 same `TaskGroup` structure, different client.\n- Callers that want per-URL failures rather than fail-fast should catch\n `ExceptionGroup` and inspect `.exceptions`, or wrap `_fetch_one` to return\n a `Result` union per URL. `except*` can destructure the group by exception\n type where that reads clearer.\n- There is no blocking call inside any `async def` here; if you swap in a\n library whose client is synchronous, route it through\n `asyncio.to_thread` rather than awaiting it directly.\n\nOn the exception contract: on the first failure the `TaskGroup` cancels the\nother in-flight requests and re-raises an `ExceptionGroup`. That is usually\nwhat you want for \"fetch all or fail\", but it means a partial result set is\ndiscarded. If the caller needs best-effort (return whatever succeeded, plus\nthe errors), say so and I'll rework `_fetch_one` to return a\n`tuple[bool, str | Exception]`-style result instead.",
1099
+ "outputSha256": "ed99bf7d36967e18a639d3d5ff671626371738d02c43daf1fd2c1958443a68a2",
1100
+ "promptSha256": "c6993d7ca7d8c2627ee85c60b53e4d5b3a07e0f1764c0f4197846488c91dd7c5",
1101
+ "deterministic": [
1102
+ true
1103
+ ],
1104
+ "judge": {
1105
+ "verdict": "pass",
1106
+ "reason": "Code uses `async with asyncio.TaskGroup() as tg:` with `tg.create_task(...)` per URL (criterion 1), fetches via `httpx.AsyncClient` with `await client.get(url)` (criterion 2), and returns `results` list after the group exits (criterion 3). No blocking `requests.get` inside the coroutine; the mention of `to_thread` is only a warning."
1107
+ },
1108
+ "passed": true
1109
+ },
1110
+ {
1111
+ "output": "I can't fully implement this without your codebase in front of me \u2014 the skill's first step is discovering the project's own tooling (`pyproject.toml`, HTTP client, conventions), and I have no file access here. But here's a complete, current-practice implementation to adapt.\n\n## Choice: `asyncio.TaskGroup` (3.11+)\n\nFor a fixed set of related awaitables, `asyncio.TaskGroup` is the current best practice: unlike `asyncio.gather`, if one task raises, the others are cancelled and the error propagates via an `ExceptionGroup` \u2014 no silent partial failure, no leaked tasks. `gather` is the fallback only if the project's `requires-python` is below 3.11.\n\n## Implementation\n\n```python\nfrom __future__ import annotations\n\nimport asyncio\nfrom dataclasses import dataclass\n\n\n@dataclass(frozen=True, slots=True)\nclass FetchResult:\n url: str\n status: int\n body: str\n\n\nasync def _fetch_one(\n client: \"HttpxClient\", # e.g. httpx.AsyncClient\n url: str,\n) -> FetchResult:\n response = await client.get(url)\n response.raise_for_status()\n return FetchResult(url=url, status=response.status_code, body=response.text)\n\n\nasync def fetch_all(\n client: \"HttpxClient\",\n urls: list[str],\n) -> list[FetchResult]:\n \"\"\"Fetch all URLs concurrently, preserving input order.\n\n Raises an ExceptionGroup if any request fails; the remaining in-flight\n requests are cancelled.\n \"\"\"\n if not urls:\n return []\n\n results: list[FetchResult | None] = [None] * len(urls)\n\n async with asyncio.TaskGroup() as tg:\n tasks = [\n tg.create_task(_fetch_one(client, url), name=f\"fetch:{url}\")\n for url in urls\n ]\n\n for i, task in enumerate(tasks):\n results[i] = task.result()\n\n return [r for r in results if r is not None]\n```\n\n## Key decisions, and what to adjust for your project\n\n- **Client injected, not created.** I pass the HTTP client in rather than constructing one inside `fetch_all`. Creating a client per call defeats connection pooling and leaks sockets; creating one here and closing it would fight a caller who wants to reuse theirs. Adjust the type to whatever async HTTP library the project uses (`httpx`, `aiohttp`, etc.).\n- **Order preserved.** `TaskGroup` gives you unordered completion, so I collect tasks into a list and index results by position \u2014 callers of `fetch_all(urls)` expect results to line up with `urls`.\n- **`TaskGroup` over `gather`.** With `gather` + `return_exceptions=True` you get a list you must inspect for exceptions; with `gather` default, one failure can leave the others' exceptions unretrieved (logged as \"never retrieved\"). `TaskGroup` cancels siblings on failure and surfaces everything in one `ExceptionGroup`. But check `requires-python`: the `class Box[T]`-style syntax and `except*` are 3.12+, while `TaskGroup` itself is 3.11+. On 3.10 or earlier, use `gather` and handle the exception-retrieval caveat explicitly.\n- **No bare `except`.** `_fetch_one` raises a specific type via `raise_for_status()`; I let it propagate rather than swallowing. `asyncio.CancelledError` is intentionally not caught \u2014 it must propagate through any cleanup so cancellation works.\n- **Empty input short-circuits** before entering the task group, since an empty `TaskGroup` is pointless overhead.\n\n## Verify against your repo\n\n```bash\nruff check . && ruff format --check . && mypy .\npytest -x -q\n```\n\nPrefix with `uv run` / `poetry run` per your `pyproject.toml`. Before finalizing, confirm: the project's existing `Optional` vs `| None` style (I used PEP 604), its docstring convention, and whether it already has an async client factory you should use instead of injecting the client. If the existing code catches HTTP errors at a boundary, raise/channel into `fetch_all`'s caller rather than adding a `except Exception` here.",
1112
+ "outputSha256": "04bf0e1b8957f5311afcf9036302ebb9715c60e15ef753692ce33233b22dda08",
1113
+ "promptSha256": "c6993d7ca7d8c2627ee85c60b53e4d5b3a07e0f1764c0f4197846488c91dd7c5",
1114
+ "deterministic": [
1115
+ true
1116
+ ],
1117
+ "judge": {
1118
+ "verdict": "pass",
1119
+ "reason": "Uses `async with asyncio.TaskGroup() as tg` with `tg.create_task(_fetch_one(...))` (criterion 1). Fetch uses async client: `response = await client.get(url)` with `HttpxClient`/`httpx.AsyncClient` (criterion 2). Returns aggregated `list[FetchResult]` built from task results after group completes (criterion 3). No blocking `requests.get` inside the coroutine."
1120
+ },
1121
+ "passed": true
1122
+ },
1123
+ {
1124
+ "output": "Short answer up front: there is not enough context in the request to write this correctly, because the right implementation depends on facts I'd have to discover first. Rather than invent them, here's what I'd need and how the function should be shaped for each case.\n\n## What I'd check before writing code\n\n1. **`pyproject.toml`** \u2014 `requires-python` decides whether `asyncio.TaskGroup` (3.11+) is even available or whether this project is pinned to something older that needs `asyncio.gather`.\n2. **The project's HTTP client** \u2014 `httpx` (`AsyncClient`), `aiohttp` (`ClientSession`), or the stdlib. The concurrency grouping is the same; the client setup is not, and reusing an existing session/client beats constructing one per call.\n3. **Neighboring modules** \u2014 for the project's PEP 604 (`str | None`) vs. `Optional[str]` convention, docstring style, and whether it already uses `from __future__ import annotations`.\n\nA correct version, assuming Python 3.11+ and `httpx` (the most common current choice), looks like this:\n\n```python\nimport asyncio\nimport httpx\n\n\nasync def fetch_all(\n urls: list[str],\n *,\n client: httpx.AsyncClient | None = None,\n timeout: float = 30.0,\n) -> list[httpx.Response]:\n \"\"\"Fetch *urls* concurrently, returning responses in input order.\n\n On the first request failure, in-flight requests are cancelled and the\n original exception is re-raised.\n \"\"\"\n owns_client = client is None\n if client is None:\n client = httpx.AsyncClient(timeout=timeout)\n\n try:\n async with asyncio.TaskGroup() as tg:\n tasks = [tg.create_task(client.get(url)) for url in urls]\n return [task.result() for task in tasks] # TaskGroup guarantees completion\n finally:\n if owns_client:\n await client.aclose()\n```\n\n## Why `asyncio.TaskGroup` and not `asyncio.gather`\n\n`asyncio.gather` is still fine when you deliberately want \"collect results, ignore exceptions, decide later,\" but it has two problems for the common case: if it's not marked `return_exceptions=True` it leaks unretrieved-exception warnings, and if it is, you get a list of mixed `Result | Exception` values that every caller must re-discriminate. `TaskGroup` gives you a structured-concurrency guarantee instead \u2014 the `async with` block does not exit until every task is done, and if any task raises, the others are cancelled and the original exception is raised (wrapped in an `ExceptionGroup` when more than one fails). That is the behavior you almost always want for \"fetch all of these.\"\n\nTwo caveats worth being explicit about:\n\n- **`TaskGroup` requires Python 3.11+.** If `requires-python` says 3.10, this breaks on the declared minimum and I'd use `asyncio.gather(..., return_exceptions=False)` instead, accepting the weaker cancellation semantics.\n- **`ExceptionGroup` is what propagates, not the bare exception**, when multiple tasks fail. If callers need to catch a specific `httpx.HTTPError`, they must either call `.split()`/`.subgroup()` on the group or the signature's contract must say so.\n\n## Things I deliberately did *not* do\n\n- **No bare `except Exception:`** around the fetches. Hiding failures as `None` results is a silent-bug generator; the caller should decide how to handle a failed fetch.\n- **No `print`**, no f-string in a logging call \u2014 if this needed request/response logging it would go through the project's logger with lazy `%s` interpolation, not string formatting.\n- **No `httpx.AsyncClient` constructed inside the task list.** One client, shared across concurrent requests, is the point; constructing per-URL defeats connection pooling.\n\n## What I'd verify before calling it done\n\n```\nruff check .\nruff format --check .\nmypy . # or pyright\npytest -x -q\n```\n\nprefixed with the project's runner (`uv run \u2026`, `poetry run \u2026`) once Step 1 tells me which one it is. The signature is fully type-hinted; the `AsyncClient` is acquired and closed via context managers (`async with` for the session, `try/finally` for the owned client); the only blocking-style hazard \u2014 an accidental synchronous call inside the coroutines \u2014 is absent because `client.get` is awaited.\n\n**The one thing I can't decide for you:** whether partial failure should fail the whole call (as written above, via `TaskGroup`) or return a per-URL `Result`/error mapping. Those are genuinely different contracts. If the caller wants \"best effort, tell me what failed,\" say so and I'd switch to `gather(..., return_exceptions=True)` plus a small result dataclass \u2014 but I would not pick that silently.",
1125
+ "outputSha256": "3ef30c9955679f197a3d99025a98cd106cb446b81a6a6dfafdf9687e6ff24e1b",
1126
+ "promptSha256": "c6993d7ca7d8c2627ee85c60b53e4d5b3a07e0f1764c0f4197846488c91dd7c5",
1127
+ "deterministic": [
1128
+ true
1129
+ ],
1130
+ "judge": {
1131
+ "verdict": "pass",
1132
+ "reason": "Code shows `async with asyncio.TaskGroup() as tg:` with `tg.create_task(client.get(url))` (criterion 1), uses `httpx.AsyncClient` and awaits `client.get` (criterion 2), and returns `[task.result() for task in tasks]` after the group completes (criterion 3). No blocking `requests.get` inside the coroutine."
1133
+ },
1134
+ "passed": true
1135
+ },
1136
+ {
1137
+ "output": "Here's an implementation following the workflow above, presented inline since I have no file access here.\n\n**Discover first (what I'd check in a real repo):** `pyproject.toml` for `requires-python`, the dependency manager (`uv`/`poetry`/`pip`), and which HTTP client is already a dependency. I'd also read a neighboring async module to match its docstring style, import grouping, and whether it uses `Optional[X]` or `X | None`. Since I can't see your project, the code below uses `X | None` (PEP 604), `httpx.AsyncClient` (the standard async HTTP client), and `asyncio.TaskGroup` (the current best practice for structured concurrency, available in 3.11+). Tell me if your project pins an older minimum or uses `aiohttp`, and I'll adjust.\n\n```python\nimport asyncio\nimport logging\nfrom dataclasses import dataclass\n\nimport httpx\n\nlogger = logging.getLogger(__name__)\n\n\n@dataclass(frozen=True, slots=True)\nclass FetchResult:\n \"\"\"Outcome of fetching one URL.\n\n Exactly one of ``text``/``error`` is set: ``error`` is None on success.\n \"\"\"\n\n url: str\n status_code: int | None = None\n text: str | None = None\n error: str | None = None\n\n @property\n def ok(self) -> bool:\n return self.error is None and self.status_code is not None \\\n and self.status_code < 400\n\n\nasync def _fetch_one(client: httpx.AsyncClient, url: str) -> FetchResult:\n try:\n response = await client.get(url)\n response.raise_for_status()\n except httpx.HTTPStatusError as exc:\n logger.warning(\"fetch %s failed with status %s\", url, exc.response.status_code)\n return FetchResult(url=url, status_code=exc.response.status_code,\n error=f\"HTTP {exc.response.status_code}\")\n except httpx.HTTPError as exc:\n # Covers connect/read/timeout errors; chain is preserved for the traceback.\n logger.warning(\"fetch %s failed: %s\", url, exc)\n return FetchResult(url=url, error=str(exc))\n\n return FetchResult(url=url, status_code=response.status_code, text=response.text)\n\n\nasync def fetch_all(urls: list[str]) -> list[FetchResult]:\n \"\"\"Fetch all URLs concurrently and return one result per URL, in input order.\n\n A failure on one URL does not cancel the others; each failure is captured\n in the corresponding ``FetchResult.error``.\n \"\"\"\n if not urls:\n return []\n\n async with httpx.AsyncClient(timeout=httpx.Timeout(10.0)) as client:\n results: list[FetchResult] = [None] * len(urls) # type: ignore[list-item]\n async with asyncio.TaskGroup() as tg:\n tasks = [tg.create_task(_fetch_one(client, url)) for url in urls]\n # TaskGroup re-raises only if a task raised; _fetch_one catches\n # everything it expects, so this line is reached normally.\n for i, task in enumerate(tasks):\n results[i] = task.result()\n return results\n```\n\n**Key choices, mapped to the workflow:**\n\n- **Concurrency grouping is `asyncio.TaskGroup`** (Step 3.4), not the older `asyncio.gather`. TaskGroup gives structured concurrency: if any child task raises, the group cancels the siblings and re-raises as an `ExceptionGroup`, so you never leak un-awaited tasks. (`gather(..., return_exceptions=True)` is the legacy alternative; TaskGroup is the current recommendation.)\n- **One client, many requests.** An `httpx.AsyncClient` is created once and shared across all tasks so it reuses one connection pool. This is also why it's wrapped in `async with` \u2014 it's a closable resource (Step 3.2), so its connections are released deterministically even if a task raises.\n- **Per-URL error handling, specific exception types** (Step 3.3): `HTTPStatusError` (a 4xx/5xx) and its parent `HTTPError` (connect/timeout/DNS) are caught individually \u2014 never a bare `except Exception`. Because `_fetch_one` handles its own expected failures, one bad URL doesn't abort the whole batch, which is usually what callers want from \"fetch all.\"\n- **Result order matches input order.** TaskGroup doesn't return results ordered for you, so I index back into a preallocated list using each task's position, preserving `urls` order rather than completion order.\n- **Type hints on every public signature** (Step 3.1), and a frozen `slots` dataclass for the result shape (Step 2) since it's a fixed set of related fields your own code constructs.\n\n**Two things I'd flag for your repo, since I can't see it:**\n\n1. **Python version.** `TaskGroup` and `except*` need 3.11+; `@dataclass(slots=True)` needs 3.10+. If `requires-python` is below that, use `asyncio.gather` and drop `slots`.\n2. **The `type: ignore` on the preallocated list** is a real, narrow suppression: it's a standard \"fill a list by index\" pattern the checker can't prove. If your project bans `type: ignore` outright, replace with `results: list[FetchResult] = []` and `results.append(task.result())` inside the `for` loop after the TaskGroup block \u2014 same order, no suppression.\n\n**Verify (Step 4), prefixing with your project's runner):**\n\n```bash\nuv run ruff check .\nuv run ruff format --check .\nuv run mypy .\nuv run pytest -x -q\n```\n\nIf this feature needs tests, that's a hand-off to `python-testing` rather than something I'd bundle into this change.",
1138
+ "outputSha256": "9684785905b5f73dc4dc10dbead5103521c31ddf7b776e771775895dd1d8c8c0",
1139
+ "promptSha256": "c6993d7ca7d8c2627ee85c60b53e4d5b3a07e0f1764c0f4197846488c91dd7c5",
1140
+ "deterministic": [
1141
+ true
1142
+ ],
1143
+ "judge": {
1144
+ "verdict": "pass",
1145
+ "reason": "Code uses `async with asyncio.TaskGroup() as tg:` with `tg.create_task(_fetch_one(client, url))` (criterion 1), fetches via `httpx.AsyncClient` with `await client.get(url)` (criterion 2), and returns `results` list of FetchResult for all URLs after the group completes (criterion 3). No blocking requests.get inside the coroutine."
1146
+ },
1147
+ "passed": true
1148
+ },
1149
+ {
1150
+ "output": "I can't write this to a file or run a checker here (no tools in this exercise), so here's the complete implementation inline, followed by what I'd verify once tools are available.\n\n## Implementation\n\n```python\nimport asyncio\nfrom typing import Final\n\nimport httpx\n\nDEFAULT_TIMEOUT: Final = 30.0\nDEFAULT_CONCURRENCY: Final = 10\n\n\nasync def fetch_all(\n urls: list[str],\n *,\n client: httpx.AsyncClient | None = None,\n timeout: float = DEFAULT_TIMEOUT,\n max_concurrency: int = DEFAULT_CONCURRENCY,\n) -> list[str]:\n \"\"\"Fetch multiple URLs concurrently and return their bodies in input order.\n\n Args:\n urls: URLs to fetch. An empty list short-circuits to ``[]``.\n client: Optional caller-owned ``httpx.AsyncClient`` to reuse. When\n ``None``, a temporary client is created and closed on exit.\n timeout: Per-request timeout in seconds.\n max_concurrency: Upper bound on in-flight requests.\n\n Returns:\n Response bodies as ``str``, positionally matching ``urls``.\n\n Raises:\n httpx.HTTPStatusError: If any response has a 4xx/5xx status.\n httpx.RequestError: If any request fails at the transport level.\n \"\"\"\n if not urls:\n return []\n\n semaphore = asyncio.Semaphore(max_concurrency)\n owns_client = client is None\n active_client = client or httpx.AsyncClient(timeout=timeout)\n\n async def fetch_one(url: str) -> str:\n async with semaphore:\n response = await active_client.get(url)\n response.raise_for_status()\n return response.text\n\n try:\n async with asyncio.TaskGroup() as tg:\n tasks = [tg.create_task(fetch_one(url)) for url in urls]\n finally:\n if owns_client:\n await active_client.aclose()\n\n return [task.result() for task in tasks]\n```\n\n## Why these choices\n\n**`asyncio.TaskGroup` (3.11+) instead of `asyncio.gather`.** This is the current recommendation for grouping related concurrent calls. The key advantage is structured concurrency: if any child task raises, `TaskGroup` cancels the remaining siblings and then re-raises the failures as an `ExceptionGroup` (a `BaseExceptionGroup` where all children are `Exception` subclasses \u2014 retrievable as an `ExceptionGroup` here). With `gather(..., return_exceptions=False)`, the first exception surfaces but the other coroutines are not cancelled; they keep running detached. The `async with asyncio.TaskGroup() as tg:` block does not exit until every child has completed or been cancelled, so there are no orphaned in-flight requests after the function returns.\n\n**One shared `AsyncClient`, not one per URL.** `httpx.AsyncClient` (and `aiohttp.ClientSession`) holds a connection pool; creating a client per request defeats keep-alive and is a common async performance bug. The `client` parameter lets callers inject a long-lived, app-scoped client so connection reuse survives across `fetch_all` calls. When we create it ourselves we close it in a `finally`, because `AsyncClient` is a closable resource and skipping the close leaks connections \u2014 this is the async analogue of the \"always use `with`\" rule.\n\n**`asyncio.Semaphore` for bounded concurrency.** Firing off hundreds of tasks against a host will get you rate-limited or connection-starved. The semaphore is acquired inside `fetch_one` (i.e. inside the task), not before creating the task \u2014 acquiring it outside would serialize task creation and defeat the point. Note that `tg.create_task` schedules the coroutine immediately, so all tasks exist; the semaphore limits how many hold a connection at once, not how many tasks are created. For very large `urls` you'd additionally batch, but that's premature here.\n\n**Ordering.** `tg.create_task` returns tasks and I collect them into a list in `urls` order, then read `task.result()` in that same order. Reading results after the `async with` block has exited is safe because every task is guaranteed done by then. This gives positional correspondence between inputs and outputs, which callers almost always want.\n\n**No `asyncio.CancelledError` swallowing.** There is no `except Exception` around the await. `CancelledError` inherits from `BaseException` in 3.8+, so a broad `except Exception` wouldn't catch it anyway, but the point stands: cancellation should propagate. If the caller of `fetch_all` is cancelled, the `TaskGroup` cancels its children and the `finally` still runs `aclose()`. I have deliberately *not* wrapped this in a `try/except` that suppresses failures \u2014 a partial result set presented as success is a worse failure mode than a raised `ExceptionGroup`.\n\n**No blocking calls in `async def`.** The only awaits are on `httpx` async methods and `aclose()`; there is no `requests.get`, `time.sleep`, or synchronous file I/O inside the coroutine.\n\n## Typing note\n\n`list[str]` and `X | None` (PEP 604) require 3.10+/3.9+-with-`from __future__ import annotations`. If the target project's `requires-python` is 3.8 or 3.9, switch to `List[str]`/`Optional[httpx.AsyncClient]` and add `from __future__ import annotations`. Before dropping this in I'd read `pyproject.toml` to confirm the project already depends on `httpx` (vs. `aiohttp`, which needs the `ClientSession` + `asyncio.gather` shape instead), check whether it uses `uv`/`poetry` so the verify commands get the right prefix, and match the docstring style of neighboring modules.\n\n## What I'd verify once I have tools\n\n- `ruff check .` and `ruff format --check .` exit 0 on the file.\n- `mypy`/`pyright` clean; in particular that `task.result()` is typed `str`, not `Any` \u2014 `Task[str]` from `create_task` guarantees this.\n- A test that a failing URL (e.g. a 500) causes `pytest.raises(ExceptionGroup)` and that the other in-flight requests are cancelled, and a test that an empty `urls` returns `[]` without creating a client.\n- Confirm the caller closes an injected `client` (the function closes only the one it created).\n\nIf tests are needed as part of this change, that's a hand-off to the `python-testing` skill rather than something I'd write under this one.",
1151
+ "outputSha256": "e3ae245f8bb5891d45dc37548ca2e51c6e521447ffbb8eadee79c265188e6d9d",
1152
+ "promptSha256": "c6993d7ca7d8c2627ee85c60b53e4d5b3a07e0f1764c0f4197846488c91dd7c5",
1153
+ "deterministic": [
1154
+ true
1155
+ ],
1156
+ "judge": {
1157
+ "verdict": "pass",
1158
+ "reason": "Code uses `async with asyncio.TaskGroup() as tg:` with `tg.create_task(fetch_one(url))` (criterion 1), fetches via `httpx.AsyncClient` with `await active_client.get(url)` (criterion 2), and returns `[task.result() for task in tasks]` in input order (criterion 3). No blocking `requests.get` inside the coroutine; it only warns against blocking calls."
1159
+ },
1160
+ "passed": true
1161
+ },
1162
+ {
1163
+ "output": "Here's a modern implementation using `asyncio.TaskGroup` (3.11+) for structured concurrency:\n\n```python\nimport asyncio\nfrom collections.abc import Sequence\n\nimport httpx\n\n\nasync def fetch_one(client: httpx.AsyncClient, url: str) -> str:\n response = await client.get(url)\n response.raise_for_status()\n return response.text\n\n\nasync def fetch_all(urls: Sequence[str]) -> list[str]:\n \"\"\"Fetch every URL concurrently, returning bodies in input order.\n\n Raises the first exception raised by any fetch; the TaskGroup cancels\n the remaining tasks once one fails.\n \"\"\"\n async with httpx.AsyncClient(timeout=10.0) as client:\n results: dict[int, str] = {}\n async with asyncio.TaskGroup() as tg:\n for i, url in enumerate(urls):\n tg.create_task(_store(results, i, fetch_one(client, url)))\n return [results[i] for i in range(len(urls))]\n\n\nasync def _store(results: dict[int, str], index: int, coro) -> None:\n results[index] = await coro\n```\n\nKey points about this implementation:\n\n**Why `TaskGroup` over `asyncio.gather`.** `TaskGroup` is the current best practice (introduced in 3.11). Unlike `gather`, it does not silently swallow sibling tasks when one fails \u2014 if any child raises, the group cancels all remaining tasks and raises an `ExceptionGroup`. It's a structured-concurrency primitive, so no task can outlive the `async with` block. If you need `gather` semantics deliberately (e.g. `return_exceptions=True` to collect per-URL errors rather than failing fast), that's a conscious choice, not a default.\n\n**Single shared client.** `httpx.AsyncClient` owns a connection pool; creating one per request defeats keep-alive and adds overhead. Reusing one client across all fetches \u2014 opened via `async with` so it's always closed \u2014 is the correct pattern.\n\n**Order preservation.** `TaskGroup` gives no completion-order guarantee, so results are written to an index-keyed dict and reassembled in input order at the end. If you don't care about order, return `list(results.values())` instead.\n\n**Error behavior.** `raise_for_status()` turns HTTP error codes into exceptions; those propagate out of the `TaskGroup` wrapped in an `ExceptionGroup`. If you'd rather return per-URL outcomes, wrap `fetch_one` in a try/except and return a result-or-error object per URL instead of letting it raise.\n\n**No blocking calls.** All I/O is awaited inside `async def`; no `requests`, `time.sleep`, or synchronous file/network calls appear in the coroutine.\n\nTwo variants worth knowing:\n\n- If your `requires-python` is below 3.11, `TaskGroup` is unavailable \u2014 fall back to `asyncio.gather(*[fetch_one(client, u) for u in urls])`, accepting its cancellation semantics, or use the `taskgroup` backport.\n- For a bounded number of concurrent requests (e.g. a rate limit), wrap the loop with an `asyncio.Semaphore` acquired inside `fetch_one`, rather than spawning one task per URL unbounded.\n\n**Verification:** before reporting this done, run the project's own `ruff check`/`ruff format --check`, `mypy`/`pyright` (the `dict[int, str]` annotation and `Sequence[str]` param should type-check cleanly), and `pytest`. Confirm the new code introduces no bare `except`, no unclosed client, and no blocking call inside `async def`. If the project must also run on 3.10 or earlier, verify `requires-python` and swap `TaskGroup` for `gather` as above.",
1164
+ "outputSha256": "dfee63c47472997713e304b5c299a2ed50b8277645b3fd00472e98ee9668e963",
1165
+ "promptSha256": "c6993d7ca7d8c2627ee85c60b53e4d5b3a07e0f1764c0f4197846488c91dd7c5",
1166
+ "deterministic": [
1167
+ true
1168
+ ],
1169
+ "judge": {
1170
+ "verdict": "pass",
1171
+ "reason": "Code uses `async with asyncio.TaskGroup() as tg:` with `tg.create_task(...)` (criterion 1), fetches via `httpx.AsyncClient` with `await client.get(url)` (criterion 2), and returns `[results[i] for i in range(len(urls))]` after the group completes (criterion 3). No blocking `requests.get` inside the coroutine."
1172
+ },
1173
+ "passed": true
1174
+ },
1175
+ {
1176
+ "output": "```python\nimport asyncio\nfrom dataclasses import dataclass\n\nimport httpx\n\n\n@dataclass(frozen=True)\nclass FetchResult:\n url: str\n status: int\n body: str\n\n\nasync def _fetch_one(\n client: httpx.AsyncClient,\n url: str,\n) -> FetchResult:\n response = await client.get(url)\n return FetchResult(url=url, status=response.status_code, body=response.text)\n\n\nasync def fetch_all(urls: list[str]) -> list[FetchResult]:\n \"\"\"Fetch every URL concurrently, preserving input order.\n\n All requests share a single connection pool and run inside a TaskGroup,\n so if any request fails the others are cancelled and the first error is\n re-raised as part of an ExceptionGroup.\n \"\"\"\n results: list[FetchResult | None] = [None] * len(urls)\n\n async with httpx.AsyncClient() as client:\n async with asyncio.TaskGroup() as tg:\n for i, url in enumerate(urls):\n results[i] = await tg.create_task(_fetch_one(client, url))\n\n return [r for r in results if r is not None]\n```\n\nNotes on the choices above, keyed to the checklist:\n\n- **`asyncio.TaskGroup` (3.11+)** is the current best practice for\n \"run these awaitables together.\" Unlike `asyncio.gather`, it guarantees\n every child task is finished (or cancelled) before the `async with` block\n exits, and it propagates an `ExceptionGroup` carrying *all* failures rather\n than silently dropping some. Requires Python 3.11+ \u2014 if the project's\n `requires-python` is lower, fall back to `asyncio.gather(*..., return_exceptions=...)`\n and document the difference.\n- **One shared `httpx.AsyncClient`** for the whole batch, opened with\n `async with` so the connection pool is closed even if a task raises. Each\n request is dispatched as its own task, so they overlap.\n- **`CancelledError` is not caught**, so when one request raises the TaskGroup\n cancels its siblings and the cleanup in the `async with` blocks still runs.\n- **Order is preserved** via the pre-sized `results` list indexed by input\n position, rather than relying on task completion order.\n- **Result typing** uses a frozen `@dataclass` rather than returning raw\n `httpx.Response` objects, so callers aren't handed a live connection tied\n to a closed client.\n\nIntegration checklist before merging:\n\n- Run `ruff check .`, `ruff format --check .`, and `mypy .` (or the project's\n `uv run`/`poetry run` equivalents) and confirm exit 0.\n- Swap `httpx` for whichever async HTTP client the project already depends\n on (e.g. `aiohttp`) rather than adding a second one.\n- If failures should be per-URL rather than all-or-nothing, catch\n `httpx.HTTPError` inside `_fetch_one` and return a result variant carrying\n the error instead of letting it bubble to the TaskGroup.\n- Add a timeout / retry policy if the service talks to untrusted endpoints \u2014\n out of scope here, but the shared client is the place to put it.",
1177
+ "outputSha256": "605298494761c8910cf8612994946b9e0e413be004dd7c655462a7984d9146a9",
1178
+ "promptSha256": "c6993d7ca7d8c2627ee85c60b53e4d5b3a07e0f1764c0f4197846488c91dd7c5",
1179
+ "deterministic": [
1180
+ true
1181
+ ],
1182
+ "judge": {
1183
+ "verdict": "pass",
1184
+ "reason": "Code uses `async with asyncio.TaskGroup() as tg` with `tg.create_task(_fetch_one(...))` (criterion 1), fetches via `httpx.AsyncClient` with `await client.get(url)` (criterion 2), and returns the aggregated `results` list after the group exits (criterion 3). No blocking `requests.get` inside the coroutine."
1185
+ },
1186
+ "passed": true
1187
+ }
1188
+ ]
1189
+ },
1190
+ {
1191
+ "id": "resource-with-block",
1192
+ "kind": "behavior",
1193
+ "prompt": "Write a Python function that opens a file, reads its JSON contents, and returns the parsed data.",
1194
+ "strictness": "high",
1195
+ "trials": 10,
1196
+ "passes": 10,
1197
+ "passRate": 1,
1198
+ "passAtK": 1,
1199
+ "grader": "judge",
1200
+ "status": "ran",
1201
+ "trialRecords": [
1202
+ {
1203
+ "output": "Here's a well-typed implementation following modern Python practice. Since you gave no existing project to match, I'll note the assumptions inline and use 3.12+ style (PEP 604 unions, built-in generics).\n\n```python\nfrom __future__ import annotations\n\nimport json\nfrom pathlib import Path\nfrom typing import Any\n\n\ndef read_json_file(path: str | Path) -> Any:\n \"\"\"Open *path*, parse it as JSON, and return the resulting object.\n\n Raises:\n FileNotFoundError: if *path* does not exist.\n IsADirectoryError: if *path* is a directory.\n json.JSONDecodeError: if the file contents are not valid JSON.\n \"\"\"\n with open(path, encoding=\"utf-8\") as f:\n return json.load(f)\n```\n\n### Notes on the choices\n\n- **`with open(...)`** \u2014 the file is closed even if `json.load` raises mid-parse. Never call `.close()` manually at the end of the function; that line is skipped the moment `json.load` raises a `JSONDecodeError`.\n- **`encoding=\"utf-8\"`** \u2014 explicit rather than relying on the platform default (`locale.getpreferredencoding()`), which differs across machines and will silently mis-decode non-ASCII content on a Windows box with a legacy code page.\n- **`Path`-or-`str` parameter** \u2014 `open()` accepts both, so annotated as `str | Path` to avoid forcing callers to convert.\n- **Return type `Any`** \u2014 `json.load` can produce any JSON value (`dict`, `list`, `str`, `int`, `float`, `bool`, `None`). If you know the expected shape, say so at the call site with a `TypedDict` or a validating step rather than casting to a concrete type here, which would lie about what the parser actually returns.\n\n### If the expected shape is known\n\n`Any` propagates unchecked into every caller. When you know the structure, model it and validate at the boundary instead of trusting the JSON:\n\n```python\nfrom typing import TypedDict, cast\n\nclass Config(TypedDict):\n host: str\n port: int\n\ndef read_config(path: str | Path) -> Config:\n with open(path, encoding=\"utf-8\") as f:\n data = json.load(f)\n if not isinstance(data, dict) or not isinstance(data.get(\"host\"), str) \\\n or not isinstance(data.get(\"port\"), int):\n raise ValueError(f\"{path}: expected a config object with host/port\")\n return cast(Config, data)\n```\n\n(The `cast` is safe only because the `isinstance` guards directly above it establish the shape; a bare `cast` without those checks would just move the failure downstream.)\n\n### A note on parsing strategy\n\n`json.load(f)` reads the whole stream and parses it in one call, which is correct for config-sized files and simplest to reason about. Two cases where you'd reach for something else:\n\n- **Very large files** \u2014 `json.load` still holds the entire text plus the parsed tree in memory. For multi-hundred-MB inputs, a streaming parser (`ijson`) avoids materializing the whole document.\n- **Untrusted input** \u2014 the stdlib `json` module is safe by design (it does not eval arbitrary code, unlike `yaml.load` or `pickle.load`), but a large or deeply-nested document can still exhaust memory or blow the recursion limit; if the source is untrusted, bound the file size before parsing.\n\nOne caveat if this is a service reading on a hot path: whatever you do here runs synchronously. Calling `read_json_file` directly inside an `async def` blocks the event loop for the duration of the I/O \u2014 offload it with `await asyncio.to_thread(read_json_file, path)` if that matters for the workload.\n\n### Verification before shipping\n\nWith the tooling you'd configure per-project (never assume `uv`/`poetry`; check `pyproject.toml` first):\n\n```\nruff check .\nruff format --check .\nmypy . # or pyright\npytest -x -q\n```\n\nFor a change this small the meaningful check is a test that feeds invalid JSON and an encoding with non-ASCII characters through the function and asserts the error type and round-tripped value \u2014 but I've kept that out of scope here rather than commit a test file you didn't ask for.",
1204
+ "outputSha256": "4fb96d7763c4b73ca38d98e6afb76c9ef0ae83126c112e1631de1c0980ce0a0d",
1205
+ "promptSha256": "ad15e5a9cd8603146cbf01c8bc1ada19684ba9b73526ae960a6cf9d144747774",
1206
+ "deterministic": [],
1207
+ "judge": {
1208
+ "verdict": "pass",
1209
+ "reason": "The answer shows `with open(path, encoding=\"utf-8\") as f: return json.load(f)`, satisfying criterion 1 (with block closes handle even if json.load raises), criterion 2 (json.load parses and returns data), and criterion 3 (actual code shown). It also explicitly warns against manual .close(), so no fail criterion holds."
1210
+ },
1211
+ "passed": true
1212
+ },
1213
+ {
1214
+ "output": "```python\nimport json\nfrom pathlib import Path\n\n\ndef read_json(path: str | Path) -> object:\n \"\"\"Read a UTF-8 JSON file and return the parsed data.\n\n Raises:\n FileNotFoundError: if *path* does not exist.\n json.JSONDecodeError: if the file contents are not valid JSON.\n \"\"\"\n with open(path, encoding=\"utf-8\") as f:\n return json.load(f)\n```\n\nNotes on the choices:\n\n- `with open(...)` guarantees the file is closed even if `json.load` raises\n (e.g. on malformed JSON), rather than leaking the handle.\n- `encoding=\"utf-8\"` is explicit \u2014 the default is locale-dependent, which is a\n common source of bugs on non-UTF-8 hosts.\n- The parameter accepts `str | Path` so callers can pass either; `open` handles\n both via `os.PathLike`.\n- I let the specific exceptions (`FileNotFoundError`, `json.JSONDecodeError`)\n propagate instead of catching broad `Exception` \u2014 the caller can decide how to\n handle a missing file versus malformed content, and those are meaningfully\n different failures.\n- Return type is `object` because JSON is untyped at the top level. If you know\n the expected shape, narrow it with `TypedDict` for an object, `list[...]` for\n an array, etc., and validate the parsed result against that shape before\n returning \u2014 `json.load` only guarantees the syntax is valid, not the schema.\n\nIf this file is read in a hot path or alongside other I/O, wrapping it with\n`asyncio.to_thread(read_json, path)` from an async caller avoids blocking the\nevent loop on the disk read.",
1215
+ "outputSha256": "50e7b59fbbab4693c60647e9fba3fe74debf3fc54c87b677807c2477aba897db",
1216
+ "promptSha256": "ad15e5a9cd8603146cbf01c8bc1ada19684ba9b73526ae960a6cf9d144747774",
1217
+ "deterministic": [],
1218
+ "judge": {
1219
+ "verdict": "pass",
1220
+ "reason": "Code shows `with open(path, encoding=\"utf-8\") as f: return json.load(f)` \u2014 a with block guarantees closure even if json.load raises (criterion 1), uses json.load and returns the parsed data (criterion 2), and the actual implementation is shown (criterion 3). No manual .close() leak."
1221
+ },
1222
+ "passed": true
1223
+ },
1224
+ {
1225
+ "output": "I'll write this following the workflow, though I have to be upfront: I have no tools in this exercise, so I can't actually read your `pyproject.toml`, check `requires-python`, or see neighbor modules' conventions (Step 1). What follows assumes a modern (3.12+) project with a standard `src/` layout and `ruff`/`mypy`; adjust the version-specific bits if your `requires-python` is lower.\n\n```python\nimport json\nfrom pathlib import Path\nfrom typing import Any\n\n\ndef load_json(path: str | Path) -> Any:\n \"\"\"Open *path*, parse its JSON contents, and return the result.\n\n Raises:\n FileNotFoundError: if *path* does not exist.\n IsADirectoryError: if *path* is a directory.\n json.JSONDecodeError: if the file contents are not valid JSON.\n \"\"\"\n with open(path, \"rb\") as f:\n return json.load(f)\n```\n\nKey decisions:\n\n**Resource management** \u2014 the file is opened with `with`, so it closes even if `json.load` raises on malformed input (Step 3.2). No `.close()` at the end.\n\n**Return type** \u2014 I chose `Any`. JSON can be an object, array, string, number, boolean, or null at the top level, so there is no single precise return type. If you know your files always contain an object, narrow it:\n\n```python\nimport json\nfrom pathlib import Path\n\ndef load_json_object(path: str | Path) -> dict[str, object]:\n with open(path, \"rb\") as f:\n data = json.load(f)\n if not isinstance(data, dict):\n raise ValueError(f\"expected a JSON object in {path!r}, got {type(data).__name__}\")\n return data\n```\n\nThat `isinstance` check is the honest fix for `Any` \u2014 it validates the assumption at the boundary instead of casting it away with `# type: ignore`, which the rules forbid.\n\n**Binary mode** \u2014 I open in `\"rb\"`. `json.load` detects the encoding per RFC 8259 (UTF-8/16/32) from the byte stream and is more robust across platforms than `open(path, \"r\")` with a locale-dependent default encoding. If you specifically need UTF-8 text mode, use `open(path, encoding=\"utf-8\")`.\n\n**Error handling** \u2014 I deliberately let `FileNotFoundError`, `IsADirectoryError`, and `json.JSONDecodeError` propagate rather than catching broadly. Each is a distinct, meaningful failure the caller should handle; a bare `except Exception` here would mask bugs elsewhere (see the Red Flags table). `JSONDecodeError` is a subclass of `ValueError` if you need to catch it as such.\n\n**Path handling** \u2014 accepting `str | Path` is the idiomatic modern signature. Note I did *not* add `from __future__ import annotations` \u2014 unnecessary on 3.12+ where `X | None` syntax is native. If your project targets 3.9, use `Optional`/`Union` and a forward-compat import instead, matching whatever neighboring modules do.\n\n**Logging** \u2014 I added no logging. This is a pure read helper; callers own the decision of what is worth logging, and a library function printing or logging on every call is noise.\n\nVerification I'd run against the real project (I can't execute these here):\n\n```bash\nruff check src/\nruff format --check src/\nmypy src/\npytest -x -q\n```\n\nI could not actually clear the \"Verification\" gate for you \u2014 I have no shell, no file access, and cannot confirm your tooling, `requires-python`, or docstring style. Before reporting this done, check the config, match your project's docstring convention (this uses Google-style `Raises:`), and confirm `ruff`/`mypy`/`pytest` pass on the touched file. If new tests are warranted, hand off to `python-testing` rather than folding them in here.",
1226
+ "outputSha256": "dc626fe91be8655f2449fe11c57bc3c8bf33040fd4289f53abaf68c38bd2a1aa",
1227
+ "promptSha256": "ad15e5a9cd8603146cbf01c8bc1ada19684ba9b73526ae960a6cf9d144747774",
1228
+ "deterministic": [],
1229
+ "judge": {
1230
+ "verdict": "pass",
1231
+ "reason": "The answer shows concrete code: `with open(path, \"rb\") as f: return json.load(f)` \u2014 a `with` block guarantees closure even if json.load raises (criterion 1), uses json.load and returns the parsed data (criterion 2), and the actual code is shown (criterion 3). No manual .close() happy-path leak; it explicitly warns against that."
1232
+ },
1233
+ "passed": true
1234
+ },
1235
+ {
1236
+ "output": "Here's the function, with the reasoning inline:\n\n```python\nimport json\nfrom pathlib import Path\nfrom typing import Any\n\n\ndef read_json_file(path: str | Path) -> Any:\n \"\"\"Read *path* and return its parsed JSON contents.\n\n Raises:\n FileNotFoundError: if *path* does not exist.\n json.JSONDecodeError: if the file's contents are not valid JSON.\n \"\"\"\n with Path(path).open(\"r\", encoding=\"utf-8\") as f:\n return json.load(f)\n```\n\nWhy it looks the way it does, per the skill:\n\n**Resource management.** `Path.open` is used as a context manager, so the file descriptor is closed even if `json.load` raises mid-read. An explicit `.close()` at the end would leak the descriptor on any exception before that line.\n\n**Encoding.** `encoding=\"utf-8\"` is stated explicitly rather than relying on the locale default, which varies by machine and would make parsing results platform-dependent.\n\n**Path handling.** Accepting `str | Path` (PEP 604 union, as in the skill's typing guidance) lets callers pass either without the function caring; `Path(path)` normalizes both. If your project's convention is `Optional[X]` instead, mirror that style.\n\n**No swallowed exceptions.** I do not wrap the body in `try`/`except Exception`. `FileNotFoundError` and `json.JSONDecodeError` are specific, meaningful failures the caller can handle; catching broadly would disguise a corrupt-config bug as a silent success. The docstring documents which exceptions escape so callers can decide.\n\n**Return type.** `Any` is honest here: a JSON document can be an object, array, string, number, bool, or null, so there is no single precise type. If you know the expected shape at the call site, prefer parsing into a `TypedDict` or a `@dataclass` and validating there rather than widening this signature \u2014 but for a generic reader, `Any` is the accurate annotation, not a shortcut around one.\n\nBefore dropping this into a real codebase, confirm against Step 1 of the workflow: check `pyproject.toml` for the declared `requires-python` (if it says 3.9, use `Union[str, Path]` or `from __future__ import annotations` instead of bare `str | Path`), and see whether a neighboring module already has a config/IO helper this should extend rather than duplicate. Then verify with the project's own runners:\n\n```bash\nruff check .\nruff format --check .\nmypy . # or pyright\npytest -x -q\n```\n\nprefixing each with `uv run` or `poetry run` if that's what the project uses.",
1237
+ "outputSha256": "3ee120b1fa2bb38c7f11c5333d1b791511fe0d7730088e0356f25bbce6a3208b",
1238
+ "promptSha256": "ad15e5a9cd8603146cbf01c8bc1ada19684ba9b73526ae960a6cf9d144747774",
1239
+ "deterministic": [],
1240
+ "judge": {
1241
+ "verdict": "pass",
1242
+ "reason": "Code uses `with Path(path).open(\"r\", encoding=\"utf-8\") as f: return json.load(f)` \u2014 a with block guaranteeing closure even if json.load raises (criterion 1), parses via json.load and returns the data (criterion 2), and shows the full concrete implementation (criterion 3). No bare open()/manual .close() leak."
1243
+ },
1244
+ "passed": true
1245
+ },
1246
+ {
1247
+ "output": "Here's a modern-Python implementation following the patterns above. I'll note the assumptions since I can't inspect your project in this exercise.\n\n```python\nfrom __future__ import annotations\n\nimport json\nfrom pathlib import Path\nfrom typing import Any\n\n\ndef read_json(path: str | Path) -> Any:\n \"\"\"Read a file and return its parsed JSON contents.\n\n Args:\n path: Path to a UTF-8 encoded JSON file.\n\n Returns:\n The parsed JSON data (a dict, list, str, int, float, bool, or None,\n depending on the document's top-level value).\n\n Raises:\n FileNotFoundError: If `path` does not exist.\n json.JSONDecodeError: If the file does not contain valid JSON.\n OSError: If the file cannot be opened or read.\n \"\"\"\n text = Path(path).read_text(encoding=\"utf-8\")\n return json.loads(text)\n```\n\nKey decisions and why:\n\n- **`str | Path` parameter, `Path` internally** \u2014 accepts both a string and a `pathlib.Path` at the call site without forcing the caller to convert, while using the modern `Path.read_text` API internally. This is PEP 604 union syntax; if your `requires-python` is 3.9 or earlier, switch to `Union[str, Path]` and drop the `from __future__ import annotations` reliance for that annotation. If it's below 3.9, also replace `| None`-style unions throughout and consider `os.PathLike` handling manually.\n\n- **`read_text(encoding=\"utf-8\")` then `json.loads`** rather than `open()` + `json.load`. `Path.read_text` handles the file open/close safely (equivalent to a `with` block) and makes the explicit UTF-8 encoding visible \u2014 omitting `encoding` would use the platform default, which is a real bug source on Windows. It also avoids leaking a file handle if parsing fails partway.\n\n- **No `try`/`except` around the read** \u2014 the natural exceptions (`FileNotFoundError`, `json.JSONDecodeError`, `OSError`) are specific and well-named, so they propagate to the caller who actually knows how to handle them. Wrapping them in a broad `except Exception` or a custom exception here would hide information without a concrete recovery path.\n\n- **`Any` return type** \u2014 JSON's top-level value is genuinely unconstrained by the format. If your callers expect a specific shape, that's the place to model it: hand the raw result to a `TypedDict`/dataclass, or use `json.loads(text, object_hook=...)` with a validator. Don't paper over the shape here with a cast.\n\n- **Docstring in Google style** \u2014 matches the majority convention; convert to NumPy/reST style if that's what the surrounding modules use.\n\nVerification, assuming a `uv`-managed project (substitute your Step 1 findings \u2014 `poetry run ...`, plain `ruff`/`mypy` if globally installed):\n\n```bash\nuv run ruff check .\nuv run ruff format --check .\nuv run mypy .\nuv run pytest -x -q\n```\n\nAnd a quick sanity check of the function itself:\n\n```python\nfrom pathlib import Path\nimport tempfile\n\nwith tempfile.TemporaryDirectory() as d:\n p = Path(d) / \"example.json\"\n p.write_text('{\"name\": \"widget\", \"count\": 3}', encoding=\"utf-8\")\n assert read_json(p) == {\"name\": \"widget\", \"count\": 3}\n assert read_json(str(p)) == {\"name\": \"widget\", \"count\": 3} # str path too\n```\n\nOne thing worth flagging: if this is going to be called with untrusted input or very large files, `read_text` loads the whole file into memory \u2014 for streaming a large JSON file you'd want `json.load(f)` with an explicit `with open(path, encoding=\"utf-8\") as f:` block instead. And if the file path is built from user input, that's a `security.mdc` concern (path traversal) worth checking before shipping. Let me know your project's actual Python version, layout, and typing/docstring conventions and I'll tighten this to match exactly.",
1248
+ "outputSha256": "77dd9a9dd2a8a041044494fa28875dc75d12ff9e8e4f97764d486a8375e18f69",
1249
+ "promptSha256": "ad15e5a9cd8603146cbf01c8bc1ada19684ba9b73526ae960a6cf9d144747774",
1250
+ "deterministic": [],
1251
+ "judge": {
1252
+ "verdict": "pass",
1253
+ "reason": "Uses Path(path).read_text(encoding='utf-8') which closes the handle internally on all paths (criterion 1), then json.loads(text) and returns the result (criterion 2). Actual code is shown in the fenced block (criterion 3). No bare open() with manual .close(), so the fail criterion does not hold."
1254
+ },
1255
+ "passed": true
1256
+ },
1257
+ {
1258
+ "output": "```python\nimport json\nfrom pathlib import Path\nfrom typing import Any, Union\n\n\ndef read_json_file(path: Union[str, Path]) -> Any:\n \"\"\"Open a file, read its JSON contents, and return the parsed data.\"\"\"\n with open(path, \"r\", encoding=\"utf-8\") as f:\n return json.load(f)\n```\n\nA few notes on the choices here:\n\n**Resource handling** \u2014 the file is opened with a `with` block, so it is closed even if `json.load` raises (malformed JSON, encoding error, etc.). Don't be tempted to do `f = open(...)` / `data = json.load(f)` / `f.close()`; the `close()` is skipped on any exception in between.\n\n**Encoding** \u2014 `encoding=\"utf-8\"` is explicit. Without it, Python uses a platform-dependent default locale encoding, which differs between Linux/macOS and Windows and is a common source of \"works on my machine\" bugs when the file contains non-ASCII characters.\n\n**Return type** \u2014 `json.load` can return any JSON value: an object (`dict`), array (`list`), string, number, boolean, or `None`. `Any` is the honest annotation for \"whatever is in the file.\" If you know the shape of your file, tighten it:\n\n```python\ndef read_config(path: Union[str, Path]) -> dict[str, Any]:\n with open(path, \"r\", encoding=\"utf-8\") as f:\n data = json.load(f)\n if not isinstance(data, dict):\n raise ValueError(f\"expected a JSON object at {path}, got {type(data).__name__}\")\n return data\n```\n\nAdding the `isinstance` guard is worth it whenever callers will immediately do `data[\"key\"]` \u2014 it turns a confusing `TypeError`/`KeyError` deeper in the call stack into a clear message at the boundary.\n\n**Path type** \u2014 accepting `str | Path` and passing it straight to `open` works because `open` accepts both. Use `Path` internally if you need to do anything else with the path (e.g., `.resolve()` for error messages, or `.with_suffix()`).\n\n**Error surface** \u2014 this deliberately lets three exceptions propagate rather than catching them, since the caller is best positioned to decide what to do:\n\n- `FileNotFoundError` (and other `OSError`s) if the file can't be opened,\n- `json.JSONDecodeError` if the contents aren't valid JSON,\n- `UnicodeDecodeError` if the bytes aren't UTF-8.\n\nDon't wrap these in a bare `except Exception` \u2014 catch the specific one you can actually handle, and if you re-raise, chain it: `raise ConfigError(...) from exc`.\n\n**If you're on an async code path**, `open`/`json.load` are blocking calls and must not be called directly inside `async def`. Offload with `await asyncio.to_thread(read_json_file, path)`, or for a large file use an async file API and feed chunks to `json.loads`.",
1259
+ "outputSha256": "5475374333180a5cb103c072f8d03937d5c83d4e70bece54426a335ff70e12be",
1260
+ "promptSha256": "ad15e5a9cd8603146cbf01c8bc1ada19684ba9b73526ae960a6cf9d144747774",
1261
+ "deterministic": [],
1262
+ "judge": {
1263
+ "verdict": "pass",
1264
+ "reason": "Code uses `with open(path, \"r\", encoding=\"utf-8\") as f: return json.load(f)` \u2014 a with block guaranteeing closure even if json.load raises (criterion 1), parses via json.load and returns the data (criterion 2), and shows actual code (criterion 3). No manual .close() happy-path pattern; the .close() mention is only a warning."
1265
+ },
1266
+ "passed": true
1267
+ },
1268
+ {
1269
+ "output": "```python\nimport json\nfrom pathlib import Path\nfrom typing import Any\n\n\ndef load_json_file(path: str | Path) -> Any:\n \"\"\"Open a file, read its JSON contents, and return the parsed data.\n\n Args:\n path: Path to the JSON file to read.\n\n Returns:\n The parsed JSON data (dict, list, str, int, float, bool, or None).\n\n Raises:\n FileNotFoundError: If no file exists at ``path``.\n json.JSONDecodeError: If the file's contents are not valid JSON.\n \"\"\"\n with open(path, encoding=\"utf-8\") as f:\n return json.load(f)\n```\n\nNotes on the choices above:\n\n- `with open(...)` guarantees the file handle is closed even if `json.load` raises, per the skill's `with`-block rule for any closable resource.\n- `encoding=\"utf-8\"` is explicit because the platform default varies and JSON is UTF-8 by definition.\n- The signature is fully type-hinted and uses PEP 604 `str | Path` rather than `Optional`/`Union` \u2014 match your project's convention if it still prefers `Union[str, Path]` on older Python.\n- Return type is `Any` because JSON's top-level value isn't fixed; if your callers always expect an object, narrow it to `dict[str, Any]` and validate the shape.\n- `open` already raises a specific `FileNotFoundError` and `json.load` a specific `json.JSONDecodeError`; no bare `except` is added, letting callers handle those explicitly rather than swallowing failures.\n\nIf the project's `requires-python` is 3.9 or earlier, drop the `|` syntax for `Optional`/`Union` from `typing`. If you need to distinguish \"file missing\" from \"bad JSON\" at the call site, catch those two exceptions there rather than expanding this function.",
1270
+ "outputSha256": "4d91fe1b04e73ca0a2527c3f2dd6588380d5ad5ec9483a0a8aeec30784ae7217",
1271
+ "promptSha256": "ad15e5a9cd8603146cbf01c8bc1ada19684ba9b73526ae960a6cf9d144747774",
1272
+ "deterministic": [],
1273
+ "judge": {
1274
+ "verdict": "pass",
1275
+ "reason": "Code uses `with open(path, encoding=\"utf-8\") as f: return json.load(f)` \u2014 a with block guarantees closure even if json.load raises (criterion 1), json.load parses and returns the data (criterion 2), and the actual implementation is shown (criterion 3). No manual .close() on happy path only, so fail criterion 1 does not hold."
1276
+ },
1277
+ "passed": true
1278
+ },
1279
+ {
1280
+ "output": "```python\nimport json\nfrom pathlib import Path\nfrom typing import Any\n\n\ndef read_json(path: str | Path) -> Any:\n \"\"\"Open `path` and return its parsed JSON contents.\n\n Raises:\n FileNotFoundError: if `path` does not exist.\n json.JSONDecodeError: if the file's contents are not valid JSON.\n \"\"\"\n with Path(path).open(encoding=\"utf-8\") as f:\n return json.load(f)\n```\n\nNotes on the choices here, since this skill requires matching the project's conventions before assuming any:\n\n- **Path type**: I used `str | Path` (PEP 604) in the signature. If the project targets Python < 3.10 or still uses `Optional`/`typing.Union` style, swap to `Union[str, Path]`. Check `requires-python` and a neighboring module first.\n- **`Path.open` vs. built-in `open`**: `Path.open` is the modern idiom and returns a context manager, so the file closes deterministically even if `json.load` raises mid-parse \u2014 that's why it's a `with` block rather than `open(...)` followed by `.close()`.\n- **`encoding=\"utf-8\"`**: always pass it explicitly; the default is locale-dependent and will silently mis-decode on systems with a non-UTF-8 default.\n- **Return type `Any`**: `json.load` returns an untyped structure (`dict`/`list`/`str`/`int`/`float`/`bool`/`None`). If the caller knows the shape, prefer narrowing it \u2014 e.g. `-> dict[str, Any]`, or a `TypedDict` for a known object shape, validated at the boundary rather than trusting the file blindly. `Any` here is a deliberate \"JSON is dynamically typed\" choice, not a shortcut around typing a shape you actually know.\n\nIf the project has a shared structured-error type or expects JSON loading wrapped with validation, that wrapping belongs in this function; without seeing the codebase I've kept it to the core parse.\n\nVerification to run before calling this done (prefixed with the project's own runner \u2014 `uv run`, `poetry run`, or neither \u2014 once you've checked `pyproject.toml`):\n\n```\nruff check .\nruff format --check .\nmypy . # or pyright\n```\n\nPassing `str` or `Path` both work; passing a directory raises `IsADirectoryError` on the `.open()` call, which propagates cleanly rather than being swallowed.",
1281
+ "outputSha256": "0c9dfc945253d7427b2b4388465ec2c442be7258b11489d9d029eab6ee20b067",
1282
+ "promptSha256": "ad15e5a9cd8603146cbf01c8bc1ada19684ba9b73526ae960a6cf9d144747774",
1283
+ "deterministic": [],
1284
+ "judge": {
1285
+ "verdict": "pass",
1286
+ "reason": "Code shows `with Path(path).open(encoding=\"utf-8\") as f: return json.load(f)` \u2014 a with block guaranteeing closure even if json.load raises (criterion 1), uses json.load and returns its result (criterion 2), and the actual implementation is shown (criterion 3). No manual .close() happy-path leak."
1287
+ },
1288
+ "passed": true
1289
+ },
1290
+ {
1291
+ "output": "Here's the function, written for a modern Python (3.12+) codebase:\n\n```python\nimport json\nfrom pathlib import Path\nfrom typing import Any\n\n\ndef read_json(path: str | Path) -> Any:\n \"\"\"Read a file and return its parsed JSON contents.\n\n Args:\n path: Path to the JSON file.\n\n Returns:\n The parsed JSON data (dict, list, str, int, float, bool, or None).\n\n Raises:\n FileNotFoundError: If the file does not exist.\n json.JSONDecodeError: If the file's contents are not valid JSON.\n OSError: If the file cannot be opened or read.\n \"\"\"\n with Path(path).open(encoding=\"utf-8\") as f:\n return json.load(f)\n```\n\nNotes on the choices:\n\n- **`with ... open(...)`** \u2014 the file handle is closed on every exit path, including if `json.load` raises a `JSONDecodeError` partway through. A trailing `.close()` would leak the handle on that error path.\n- **`encoding=\"utf-8\"`** \u2014 explicit rather than relying on the locale default, which differs across platforms; JSON is UTF-8 by spec.\n- **`Path(path)`** \u2014 accepts both `str` and `Path` from callers without branching.\n- **No `try`/`except` around the `open`/`load`** \u2014 `FileNotFoundError`, `OSError`, and `json.JSONDecodeError` already carry precise, actionable information. Swallowing them into a generic error loses the distinction between \"missing file\" and \"malformed JSON\", which callers usually want to handle differently. Let them propagate and let the caller decide; if a caller needs a domain-specific error, it should catch these specific types and chain with `raise MyError(...) from exc`.\n- **Return type `Any`** \u2014 matching `json.load`'s own signature. If the call site knows the expected shape (e.g. always a `dict`), narrow the return type to a `TypedDict` for that shape instead of `Any`:\n\n```python\nfrom typing import TypedDict\n\n\nclass UserRecord(TypedDict):\n id: int\n name: str\n\n\ndef read_user(path: str | Path) -> UserRecord:\n with Path(path).open(encoding=\"utf-8\") as f:\n return json.load(f) # type: ignore[no-any-return] # trusted input shape\n```\n\nThat last variant's `type: ignore` is the one case where a suppression is warranted \u2014 the narrowing is a deliberate, documented assertion about data the type checker cannot see. If the input shape is untrusted, validate it explicitly (e.g. with `pydantic` or a manual check) rather than asserting the type.",
1292
+ "outputSha256": "4407a6c2ebeab34f6b7fdf77886fbade6a45e962de248c72ea62db4a27976c5f",
1293
+ "promptSha256": "ad15e5a9cd8603146cbf01c8bc1ada19684ba9b73526ae960a6cf9d144747774",
1294
+ "deterministic": [],
1295
+ "judge": {
1296
+ "verdict": "pass",
1297
+ "reason": "Code uses `with Path(path).open(encoding=\"utf-8\") as f: return json.load(f)` \u2014 a with block guaranteeing closure on every path including JSONDecodeError, and json.load returns parsed data. Actual code is shown, not just described. No manual .close() anti-pattern."
1298
+ },
1299
+ "passed": true
1300
+ },
1301
+ {
1302
+ "output": "```python\nimport json\nfrom pathlib import Path\nfrom typing import Any\n\n\ndef read_json(path: str | Path) -> Any:\n \"\"\"Read and parse a JSON file.\n\n Args:\n path: Path to the JSON file.\n\n Returns:\n The parsed JSON contents.\n\n Raises:\n FileNotFoundError: If the file does not exist.\n json.JSONDecodeError: If the file is not valid JSON.\n \"\"\"\n with Path(path).open(encoding=\"utf-8\") as f:\n return json.load(f)\n```\n\nNotes on the choices:\n\n- **`with` block** \u2014 the file handle is closed on every exit path, including when `json.load` raises on malformed input. A bare `open()` with a later `.close()` would leak the handle in that case.\n- **`Path(path).open()`** over `open(path)` \u2014 takes both `str` and `Path` inputs and matches modern code; falls back to plain `open()` if the project targets a style that avoids `pathlib`.\n- **`encoding=\"utf-8\"`** \u2014 explicit, so behavior doesn't depend on the platform's default locale encoding (which is not UTF-8 on Windows).\n- **`Any` return** \u2014 JSON has no single static type; `Any` is honest here. If the project knows the expected shape, tighten it in two ways instead:\n - a `TypedDict` when it must remain a `dict`, e.g. `def read_json(path: str | Path) -> Config: ...`\n - a `@dataclass` + `dataclasses`/`pydantic` `from_dict` when you want a typed object.\n- **No broad `except`** \u2014 `FileNotFoundError` and `json.JSONDecodeError` are specific and propagate unchanged; the caller decides whether either is recoverable. Don't wrap them in a catch-all.\n- Match the project's own conventions: if it uses `Optional`/`typing` heavily, `str | Path` may become `Union[str, Path]`; if it's pre-3.10, use `Union` and `Optional` since `X | None` in annotations needs `from __future__ import annotations`.\n\nIf this is a hot path or the file is untrusted/large, add a size guard or use `json.load` from a streamed reader rather than reading it all eagerly \u2014 but for the common case above, that's the whole function.",
1303
+ "outputSha256": "3969df48089cfb47e07441388d8cf306cb9ddd5bb7e9e0c5e0c94cccfb976110",
1304
+ "promptSha256": "ad15e5a9cd8603146cbf01c8bc1ada19684ba9b73526ae960a6cf9d144747774",
1305
+ "deterministic": [],
1306
+ "judge": {
1307
+ "verdict": "pass",
1308
+ "reason": "Code uses `with Path(path).open(encoding=\"utf-8\") as f: return json.load(f)` \u2014 a with block closing the handle on every path including parse errors (criterion 1), parses via json.load and returns the data (criterion 2), and shows the actual implementation (criterion 3). No manual .close() happy-path leak."
1309
+ },
1310
+ "passed": true
1311
+ }
1312
+ ]
1313
+ }
1314
+ ],
1315
+ "verdict": "pass",
1316
+ "scope": "bundled",
1317
+ "skillDigest": "3dc9253b9991cf85ff9bbb9fb29b01e86b7c8d806fc867b05cc301b1748ea914",
1318
+ "catalogDigest": "4f4016d410837e394a27e5b247e38ef2f57a1ee0baba4436ce7d3d71e223333d",
1319
+ "judgePromptVersion": "2026-09-25.1",
1320
+ "runner": "deepseek",
1321
+ "model": "deepseek-chat",
1322
+ "runnerPromptVersion": "2026-09-25.1",
1323
+ "recordedAt": "2026-09-25T05:16:23.290Z",
1324
+ "judge": "deepseek",
1325
+ "judgeModel": "deepseek-chat"
1326
+ },
1327
+ {
1328
+ "schemaVersion": "1.0.0",
1329
+ "skillId": "python/python-testing",
1330
+ "strictness": "high",
1331
+ "trials": 10,
1332
+ "triggerAccuracy": {
1333
+ "truePositive": 6,
1334
+ "falsePositive": 0,
1335
+ "positives": 6,
1336
+ "negatives": 6
1337
+ },
1338
+ "evidence": "authored",
1339
+ "scenarios": [
1340
+ {
1341
+ "id": "trigger-positive-1",
1342
+ "kind": "trigger-positive",
1343
+ "prompt": "Write pytest tests for the new widgets module",
1344
+ "strictness": "high",
1345
+ "trials": 1,
1346
+ "passes": 1,
1347
+ "passRate": 1,
1348
+ "passAtK": 1,
1349
+ "grader": "trigger-rank-fork-family",
1350
+ "status": "ran",
1351
+ "deterministic": true
1352
+ },
1353
+ {
1354
+ "id": "trigger-positive-2",
1355
+ "kind": "trigger-positive",
1356
+ "prompt": "Add test coverage for this Python function I just wrote",
1357
+ "strictness": "high",
1358
+ "trials": 1,
1359
+ "passes": 1,
1360
+ "passRate": 1,
1361
+ "passAtK": 1,
1362
+ "grader": "trigger-rank-fork-family",
1363
+ "status": "ran",
1364
+ "deterministic": true
1365
+ },
1366
+ {
1367
+ "id": "trigger-positive-3",
1368
+ "kind": "trigger-positive",
1369
+ "prompt": "Fix this failing pytest test in test_client.py",
1370
+ "strictness": "high",
1371
+ "trials": 1,
1372
+ "passes": 1,
1373
+ "passRate": 1,
1374
+ "passAtK": 1,
1375
+ "grader": "trigger-rank-fork-family",
1376
+ "status": "ran",
1377
+ "deterministic": true
1378
+ },
1379
+ {
1380
+ "id": "trigger-positive-4",
1381
+ "kind": "trigger-positive",
1382
+ "prompt": "Add a pytest fixture for the database connection used across these tests",
1383
+ "strictness": "high",
1384
+ "trials": 1,
1385
+ "passes": 1,
1386
+ "passRate": 1,
1387
+ "passAtK": 1,
1388
+ "grader": "trigger-rank-fork-family",
1389
+ "status": "ran",
1390
+ "deterministic": true
1391
+ },
1392
+ {
1393
+ "id": "trigger-positive-5",
1394
+ "kind": "trigger-positive",
1395
+ "prompt": "Parametrize this pytest test to cover empty, None, and negative inputs",
1396
+ "strictness": "high",
1397
+ "trials": 1,
1398
+ "passes": 1,
1399
+ "passRate": 1,
1400
+ "passAtK": 1,
1401
+ "grader": "trigger-rank-fork-family",
1402
+ "status": "ran",
1403
+ "deterministic": true
1404
+ },
1405
+ {
1406
+ "id": "trigger-positive-6",
1407
+ "kind": "trigger-positive",
1408
+ "prompt": "Fix this failing pytest test that forgot pytest.mark.asyncio for the coroutine",
1409
+ "strictness": "high",
1410
+ "trials": 1,
1411
+ "passes": 1,
1412
+ "passRate": 1,
1413
+ "passAtK": 1,
1414
+ "grader": "trigger-rank-fork-family",
1415
+ "status": "ran",
1416
+ "deterministic": true
1417
+ },
1418
+ {
1419
+ "id": "trigger-negative-1",
1420
+ "kind": "trigger-negative",
1421
+ "prompt": "Implement the feature this test suite will eventually cover",
1422
+ "strictness": "high",
1423
+ "trials": 1,
1424
+ "passes": 1,
1425
+ "passRate": 1,
1426
+ "passAtK": 1,
1427
+ "grader": "trigger-rank-fork-family",
1428
+ "status": "ran",
1429
+ "deterministic": true
1430
+ },
1431
+ {
1432
+ "id": "trigger-negative-2",
1433
+ "kind": "trigger-negative",
1434
+ "prompt": "Review this Python pull request for mutable default arguments and broad except clauses",
1435
+ "strictness": "high",
1436
+ "trials": 1,
1437
+ "passes": 1,
1438
+ "passRate": 1,
1439
+ "passAtK": 1,
1440
+ "grader": "trigger-rank-fork-family",
1441
+ "status": "ran",
1442
+ "deterministic": true
1443
+ },
1444
+ {
1445
+ "id": "trigger-negative-3",
1446
+ "kind": "trigger-negative",
1447
+ "prompt": "Fix this ModuleNotFoundError breaking pytest collection",
1448
+ "strictness": "high",
1449
+ "trials": 1,
1450
+ "passes": 1,
1451
+ "passRate": 1,
1452
+ "passAtK": 1,
1453
+ "grader": "trigger-rank-fork-family",
1454
+ "status": "ran",
1455
+ "deterministic": true
1456
+ },
1457
+ {
1458
+ "id": "trigger-negative-4",
1459
+ "kind": "trigger-negative",
1460
+ "prompt": "Write Jest tests for this TypeScript React component",
1461
+ "strictness": "high",
1462
+ "trials": 1,
1463
+ "passes": 1,
1464
+ "passRate": 1,
1465
+ "passAtK": 1,
1466
+ "grader": "trigger-rank-fork-family",
1467
+ "status": "ran",
1468
+ "deterministic": true
1469
+ },
1470
+ {
1471
+ "id": "trigger-negative-5",
1472
+ "kind": "trigger-negative",
1473
+ "prompt": "Run a generic quality/test-gen pass across the whole repository",
1474
+ "strictness": "high",
1475
+ "trials": 1,
1476
+ "passes": 1,
1477
+ "passRate": 1,
1478
+ "passAtK": 1,
1479
+ "grader": "trigger-rank-fork-family",
1480
+ "status": "ran",
1481
+ "deterministic": true
1482
+ },
1483
+ {
1484
+ "id": "trigger-negative-6",
1485
+ "kind": "trigger-negative",
1486
+ "prompt": "Review this Python diff for security and resource-leak issues",
1487
+ "strictness": "high",
1488
+ "trials": 1,
1489
+ "passes": 1,
1490
+ "passRate": 1,
1491
+ "passAtK": 1,
1492
+ "grader": "trigger-rank-fork-family",
1493
+ "status": "ran",
1494
+ "deterministic": true
1495
+ },
1496
+ {
1497
+ "id": "fixture-scope-guidance",
1498
+ "kind": "behavior",
1499
+ "prompt": "I need a pytest fixture used by only one test file. Where should it live and what scope should it have?",
1500
+ "strictness": "high",
1501
+ "trials": 10,
1502
+ "passes": 10,
1503
+ "passRate": 1,
1504
+ "passAtK": 1,
1505
+ "grader": "judge",
1506
+ "status": "ran",
1507
+ "trialRecords": [
1508
+ {
1509
+ "output": "A fixture used by only one test file should live directly in that test file, not in any `conftest.py`. `conftest.py` is for fixtures shared across multiple files in a directory \u2014 promote the fixture there only once a second test file actually needs it, and even then place it in the narrowest `conftest.py` that covers every file that uses it (the shared directory, not the suite root).\n\nFor scope, default to `function` (pytest's default) unless you have a concrete reason to widen it. Widening to `module` or `session` purely for speed/convenience is a common trap: a wider-scoped fixture that mutates state will leak that state into tests that assumed fresh isolation, producing order-dependent failures. Only widen scope when the setup is genuinely expensive, immutable for the duration, and safe to share \u2014 and even then, prefer reproducing the sharing need explicitly.\n\nOne related distinction: if only the *inputs* to the test vary, that's a job for `@pytest.mark.parametrize` on the test, not a fixture. Use `pytest.fixture(params=[...])` only when the setup/teardown logic itself varies per case.\n\nSo the short answer: define it inline in the single test file that needs it, at the default `function` scope, and let `conftest.py` enter the picture only when sharing actually appears.",
1510
+ "outputSha256": "8ae4ab5b7e60b24c15c3669b0fb654905ebf87dd1a442c0de4e7d6cfa56c5a06",
1511
+ "promptSha256": "ba4137e4d95f58775850cffdd5d7f85cb9cf1dfd2be627de90253a367333ec28",
1512
+ "deterministic": [],
1513
+ "judge": {
1514
+ "verdict": "pass",
1515
+ "reason": "Answer says fixture should live directly in that test file, not any conftest.py, since only one file uses it. Names default `function` scope explicitly and warns against module/session. States promotion to conftest.py (narrowest covering every file) happens only once a second test file needs it. No fail criterion met."
1516
+ },
1517
+ "passed": true
1518
+ },
1519
+ {
1520
+ "output": "A fixture used by only one test file belongs **directly in that test file**, not in any `conftest.py`. Don't promote it to a `conftest.py` until a second test file needs the same fixture \u2014 and even then, put it in the narrowest `conftest.py` that covers every file that needs it (the shared directory, not the suite root).\n\nScope: use pytest's default, **`function`**, unless you have a specific, justified reason to widen it. Each test gets a fresh instance, so tests stay isolated. Widening to `module` or `session` is tempting for speed but leaks mutable state between tests that assumed isolation \u2014 only widen it for a genuinely read-only, expensive-to-construct resource, and say why in the fixture.\n\nIf only the *inputs* vary across cases, prefer `@pytest.mark.parametrize` on the test itself over a params fixture \u2014 reserve something like `pytest.fixture(params=[...])` for when the setup/teardown logic itself varies.\n\nSo: inline in the single test file, `function` scope, promoted only when a second file needs it.",
1521
+ "outputSha256": "fa558fe2a7f66710747aa6c89cb326b59ce2dfd4f130829b1497d4bf56c98c6d",
1522
+ "promptSha256": "ba4137e4d95f58775850cffdd5d7f85cb9cf1dfd2be627de90253a367333ec28",
1523
+ "deterministic": [],
1524
+ "judge": {
1525
+ "verdict": "pass",
1526
+ "reason": "Answer states fixture belongs 'directly in that test file, not in any conftest.py' since only one file uses it (crit 1). Names default 'function' scope explicitly, warning against module/session (crit 2). Says promote to narrowest conftest.py 'until a second test file needs the same fixture' (crit 3). No root-conftest recommendation."
1527
+ },
1528
+ "passed": true
1529
+ },
1530
+ {
1531
+ "output": "Define it directly in that test file (the `test_*.py` module itself) \u2014 not in a `conftest.py`. A `conftest.py` is only warranted once a second test file needs the same fixture, and even then it goes in the narrowest `conftest.py` covering every file that needs it, not the suite root.\n\nScope: use the default `function` scope unless there's a concrete reason to widen it. `module` or `session` scope is tempting for perceived speed, but a wider-scoped fixture that mutates or accumulates state leaks between tests that assumed isolation \u2014 each test seeing a fixture already touched by a prior test produces order-dependent failures that are painful to debug. Widen the scope only when the fixture is genuinely immutable across tests (e.g. an expensive read-only resource) and the test file is the only consumer, in which case `module` is the widest that makes sense for a fixture used by one file.\n\nOne clarification on the earlier point: if you need input variations rather than different setup/teardown logic, that's `@pytest.mark.parametrize` on the test, not a wider fixture scope \u2014 keep those two axes separate:\n\n```python\n# in tests/test_parser.py \u2014 no conftest.py needed\n\n@pytest.fixture\ndef sample_config(tmp_path):\n cfg = tmp_path / \"config.toml\"\n cfg.write_text(\"[parser]\\nstrict = true\\n\")\n return cfg\n\n@pytest.mark.parametrize(\"input_val,expected\", [\n (\"a\", 1),\n (\"\", 0),\n (None, 0),\n], ids=[\"nonempty\", \"empty\", \"none\"])\ndef test_count(sample_config, input_val, expected):\n assert count(sample_config, input_val) == expected\n```\n\nThere `sample_config` stays local and function-scoped, while the varying inputs live on the test via `parametrize` \u2014 which is the right division: parametrize the test when only the inputs vary, use a fixture when the setup/teardown logic itself varies.",
1532
+ "outputSha256": "3b07b1114c36b0a25cbb6619ea7caaf38d570e6572ef0d43952547a91a1a56f7",
1533
+ "promptSha256": "ba4137e4d95f58775850cffdd5d7f85cb9cf1dfd2be627de90253a367333ec28",
1534
+ "deterministic": [],
1535
+ "judge": {
1536
+ "verdict": "pass",
1537
+ "reason": "Answer says define fixture directly in the test file, not conftest.py, since only one file uses it; recommends default function scope explicitly (bare @pytest.fixture, 'stays local and function-scoped'); states conftest promotion happens only once a second test file needs it, in the narrowest conftest covering all files. No root-conftest recommendation."
1538
+ },
1539
+ "passed": true
1540
+ },
1541
+ {
1542
+ "output": "A pytest fixture used by only one test file should live directly in that test file \u2014 not in any `conftest.py`. `conftest.py` exists so that fixtures (and hooks) are discovered across multiple test files in the same directory tree; putting a single-use fixture there widens its visible scope and discoverability for no benefit, and invites someone else to accidentally depend on it. Promote it to the narrowest `conftest.py` that covers every file needing it only once a second test file actually needs the same fixture.\n\nScope should default to `function` (pytest's own default) unless you have a concrete, measured reason to widen it. A `function`-scoped fixture is set up and torn down per test, so no state leaks between tests. Widening to `module` or `session` is a convenience optimization that mutates state shared across tests that assumed isolation \u2014 reach for it only when the fixture is genuinely expensive to construct (e.g. a DB connection, a large fixture dataset) and is provably read-only or otherwise safe to share. If setup/teardown logic itself varies across cases, that is a signal to parametrize the fixture or split it, not to widen its scope.",
1543
+ "outputSha256": "afd845f82ba87f4e954babd1a5dfb6cd8ba5fe41f66c5ca60ae70b41366129d8",
1544
+ "promptSha256": "ba4137e4d95f58775850cffdd5d7f85cb9cf1dfd2be627de90253a367333ec28",
1545
+ "deterministic": [],
1546
+ "judge": {
1547
+ "verdict": "pass",
1548
+ "reason": "Answer says fixture should live directly in that test file, not any conftest.py, because only one file uses it; explicitly recommends default `function` scope (\"Scope should default to `function`\") and warns against module/session; states promotion to narrowest conftest.py happens only once a second test file needs the same fixture. No fail criterion met."
1549
+ },
1550
+ "passed": true
1551
+ },
1552
+ {
1553
+ "output": "Put it directly in that test file, not in any `conftest.py`. A `conftest.py` is only worth creating (or extending) once a fixture is actually needed by more than one test file \u2014 and even then, put it in the narrowest `conftest.py` that covers every file needing it, not the suite root.\n\nScope: leave it at the pytest default, `function`, unless you have a concrete reason to widen it. The default gives each test a fresh fixture instance, so state mutated by one test can't leak into the next.\n\nOnly widen to `module` or `session` when the fixture does something genuinely expensive or effectful that you've decided is safe to share \u2014 spinning up a DB connection pool, for example. Widening scope \"for convenience\" is the classic way a fixture that mutates state silently couples tests that were written assuming isolation, and the resulting failures are order-dependent and miserable to debug.\n\nOne related note: if the only thing varying across your tests is the *inputs*, don't reach for a fixture at all \u2014 use `@pytest.mark.parametrize` on the test. Use a fixture when the *setup/teardown logic itself* differs, or when the resource needs lifecycle management.",
1554
+ "outputSha256": "c9196fcc69a37a0266b3b1710d6b6ce6d35db162f886a68a3ccbb26d65568d06",
1555
+ "promptSha256": "ba4137e4d95f58775850cffdd5d7f85cb9cf1dfd2be627de90253a367333ec28",
1556
+ "deterministic": [],
1557
+ "judge": {
1558
+ "verdict": "pass",
1559
+ "reason": "Answer says put fixture directly in that test file, not any conftest.py, since only one file uses it. Names default scope explicitly: 'leave it at the pytest default, `function`'. States conftest promotion happens only once needed by more than one test file, in the narrowest conftest covering all files. No root-conftest recommendation."
1560
+ },
1561
+ "passed": true
1562
+ },
1563
+ {
1564
+ "output": "Put it directly in that test file \u2014 not in any `conftest.py`. Since only one file needs it, there's no reason to widen its visible scope; promote it to the narrowest `conftest.py` that covers every file needing it only once a second test file actually needs the same fixture.\n\nFor scope, use the pytest default: `function`. That gives each test a fresh instance, so state can't leak between tests. Only widen to `module` or `session` if the fixture's setup/teardown is genuinely expensive and provably safe to share \u2014 a wider-scoped fixture that mutates state leaks between tests that assumed isolation, which produces order-dependent failures.\n\nIf what actually varies across cases is just the input values, keep the fixture at `function` scope and vary the inputs with `@pytest.mark.parametrize` on the test instead. Use `pytest.fixture(params=[...])` only when the setup/teardown logic itself varies, not merely the data.",
1565
+ "outputSha256": "6cbc1af54a3b39942bcc278f1538790716d74f650c87f10cf6b57988e54228f8",
1566
+ "promptSha256": "ba4137e4d95f58775850cffdd5d7f85cb9cf1dfd2be627de90253a367333ec28",
1567
+ "deterministic": [],
1568
+ "judge": {
1569
+ "verdict": "pass",
1570
+ "reason": "Answer says put fixture directly in that test file, not any conftest.py, since only one file needs it. Names scope explicitly: 'use the pytest default: `function`'. States promotion to narrowest conftest.py covering every file happens only once a second test file needs it. No root-conftest recommendation."
1571
+ },
1572
+ "passed": true
1573
+ },
1574
+ {
1575
+ "output": "Put it directly in that one test file, not in a `conftest.py`. `conftest.py` is for fixtures shared across multiple test files; a fixture used by a single file has no reason to be discovered by every sibling module. Promote it to the narrowest `conftest.py` that covers all files needing it only once a second test file needs it.\n\nFor scope, default to `function` (pytest's default) unless there is a measured reason to widen it. `function` gives each test a fresh instance and guarantees isolation. Widen to `module` only if the fixture is genuinely expensive and safe to share across every test in that file, and to `session` only if it is expensive, read-only, and safe for the whole run. A wider scope that mutates state leaks between tests that assumed isolation, which produces order-dependent failures.\n\nSo: inline in the test file, `function` scope. If only the inputs vary between tests, use `@pytest.mark.parametrize` on the test rather than making the fixture itself parameterized.",
1576
+ "outputSha256": "3cd5f0ecabb368fa56fe2470cdda22d6640eaad6239a3d946877dba6b569902d",
1577
+ "promptSha256": "ba4137e4d95f58775850cffdd5d7f85cb9cf1dfd2be627de90253a367333ec28",
1578
+ "deterministic": [],
1579
+ "judge": {
1580
+ "verdict": "pass",
1581
+ "reason": "Answer says put fixture directly in that one test file, not conftest.py, since only one file uses it; recommends default `function` scope explicitly; states promotion to narrowest conftest.py covering all files happens only once a second test file needs it. No root-conftest recommendation."
1582
+ },
1583
+ "passed": true
1584
+ },
1585
+ {
1586
+ "output": "A fixture used by only one test file belongs **directly in that test file itself** \u2014 not in any `conftest.py`. Don't put it in `conftest.py` \"just in case\" a second file needs it later; promote it to the narrowest `conftest.py` that covers every file needing it only once a second test file actually needs the same fixture. That keeps a single-use fixture's scope and discoverability local to where it's used.\n\nOn scope: keep the pytest default, `function` scope, unless you have a concrete reason not to. A `function`-scoped fixture gives each test a fresh instance, so one test's mutation can't leak into the next. Widening to `module` or `session` for convenience is the usual trap \u2014 a wider-scoped fixture that carries mutable state breaks the isolation the other tests assumed they had. Only widen scope when the fixture's setup is genuinely expensive or genuinely stateless, and even then verify nothing mutates it.\n\nA concrete shape:\n\n```python\n# tests/test_invoice.py\n\nimport pytest\n\n@pytest.fixture\ndef sample_invoice():\n return Invoice(lines=[Line(\"widget\", qty=2, unit_price=500)])\n\ndef test_invoice_total(sample_invoice):\n assert sample_invoice.total == 1000\n```\n\nIf instead the *setup logic* varies per case (not just the inputs), prefer `pytest.fixture(params=[...])`; if only the input *values* vary, keep the fixture and put `@pytest.mark.parametrize` on the test.",
1587
+ "outputSha256": "63b0b7bd745bf79089799c82e47240b3407c920968ff1e2768304c8d695bb46e",
1588
+ "promptSha256": "ba4137e4d95f58775850cffdd5d7f85cb9cf1dfd2be627de90253a367333ec28",
1589
+ "deterministic": [],
1590
+ "judge": {
1591
+ "verdict": "pass",
1592
+ "reason": "Answer says fixture belongs directly in that test file, not any conftest.py, since only one file uses it. It explicitly names default `function` scope (\"keep the pytest default, `function` scope\") and warns against module/session. It states promotion to the narrowest conftest.py covering every file happens only once a second test file needs it. No root-conftest recommendation."
1593
+ },
1594
+ "passed": true
1595
+ },
1596
+ {
1597
+ "output": "Put it directly in that test file \u2014 not in any `conftest.py`. A fixture needed by only one test module belongs in the module that uses it. Only promote it to a `conftest.py` once a second test file needs the same fixture, and then place it in the narrowest `conftest.py` that covers every file needing it (the shared directory, not the suite root).\n\nScope: leave it at pytest's default, `function`. Don't widen to `module` or `session` just for convenience \u2014 a wider-scoped fixture that mutates state leaks between tests that assumed isolation. Widen the scope only when the setup/teardown is genuinely expensive and provably non-mutating across the tests that consume it.\n\nIf what actually varies across tests is only the inputs rather than the setup/teardown, you don't want a fixture at all \u2014 use `@pytest.mark.parametrize` on the test. Reserve `pytest.fixture(params=[...])` for when the setup/teardown logic itself varies, not just the values fed into it.",
1598
+ "outputSha256": "e287d0261080a23dbc6532b85878a360bde3fa76e895b7efc266e40a6594a58d",
1599
+ "promptSha256": "ba4137e4d95f58775850cffdd5d7f85cb9cf1dfd2be627de90253a367333ec28",
1600
+ "deterministic": [],
1601
+ "judge": {
1602
+ "verdict": "pass",
1603
+ "reason": "Answer says put fixture directly in that test file, not any conftest.py, since only one module uses it (criterion 1). Names default scope explicitly: 'leave it at pytest's default, `function`' and warns against module/session (criterion 2). States promotion to conftest.py happens only once a second test file needs it, in the narrowest conftest covering all files (criterion 3). No fail criterion."
1604
+ },
1605
+ "passed": true
1606
+ },
1607
+ {
1608
+ "output": "A fixture used by only one test file should live directly in that test file, not in any `conftest.py`. A `conftest.py` exists for fixtures shared across multiple test files \u2014 putting a single-use fixture there widens its visible scope and discoverability for no benefit. You'd only promote it to a `conftest.py` once a second test file actually needs it, and then in the narrowest `conftest.py` that covers every file needing it (the shared directory, not the suite root).\n\nFor scope, use the natural pytest default of `function` scope. Don't widen to `module` or `session` just for convenience or speed \u2014 a wider-scoped fixture that mutates any state leaks between tests that assumed they were isolated, producing order-dependent failures. Only widen scope deliberately when the fixture is genuinely expensive to set up and provably free of mutable shared state.",
1609
+ "outputSha256": "89948fd65fbbac4714c888d56dde595d916be63b39dff9d09f53dbc6d762c0e2",
1610
+ "promptSha256": "ba4137e4d95f58775850cffdd5d7f85cb9cf1dfd2be627de90253a367333ec28",
1611
+ "deterministic": [],
1612
+ "judge": {
1613
+ "verdict": "pass",
1614
+ "reason": "Answer says fixture should live directly in that test file, not any conftest.py, since only one file uses it. Names default `function` scope explicitly, warns against module/session. States promotion to conftest.py (narrowest covering every file needing it) happens only once a second test file needs it. No fail criterion."
1615
+ },
1616
+ "passed": true
1617
+ }
1618
+ ]
1619
+ },
1620
+ {
1621
+ "id": "mock-external-not-internal",
1622
+ "kind": "behavior",
1623
+ "prompt": "Write a pytest test for a function that calls an internal helper in the same package and also calls an external HTTP API. What should be mocked?",
1624
+ "strictness": "high",
1625
+ "trials": 10,
1626
+ "passes": 10,
1627
+ "passRate": 1,
1628
+ "passAtK": 1,
1629
+ "grader": "judge",
1630
+ "status": "ran",
1631
+ "trialRecords": [
1632
+ {
1633
+ "output": "Mock the external HTTP API only \u2014 never the internal helper.\n\nConcretely:\n\n- The external HTTP call is a boundary the test doesn't own and can't rely on (network flakiness, rate limits, credentials, cost). Patch it at the point of use in the module under test, e.g. `mocker.patch(\"mypkg.mod.requests.get\")` or, better, patch whatever thin wrapper the module calls (`mocker.patch(\"mypkg.mod.http_client.fetch\")`). If the package already wraps its HTTP calls, patch the wrapper \u2014 that keeps the mock surface narrow and stable.\n- The internal helper stays real. It's part of the unit under test. Mocking it away means the test only verifies that your mocks agree with each other, and it silently breaks the moment the helper's contract changes \u2014 exactly the regression the test should catch.\n\nIf the internal helper itself performs the HTTP call, the practical shape is:\n\n```python\ndef test_get_user(mocker):\n mocker.patch(\"mypkg.mod.requests.get\", return_value=FakeResponse({\"id\": 1}))\n result = mypkg.mod.get_user(1)\n assert result[\"id\"] == 1\n # helper ran for real; only its network dependency was faked\n```\n\nA few rules that follow from this:\n\n- Patch where it's looked up (`mypkg.mod.<name>`), not where it's defined (`requests.get`), or the patch won't take effect.\n- Don't reach for `monkeypatch.setattr` on the helper \"to simplify\" \u2014 same problem as mocking it.\n- If the helper is genuinely slow or impure for a reason unrelated to the behavior under test (e.g. it touches the filesystem), that's a boundary too, and you patch *its* external dependency, not the helper itself.\n- Prefer the module's own HTTP wrapper over `requests`/`httpx` directly if one exists \u2014 it's the narrower, more honest seam.\n\nSo: one mock, on the external API's point of use. Zero mocks on the internal helper.",
1634
+ "outputSha256": "7d4ff452dcd83e035ee9989cadc0e6dae5ec96b246abfccd3cf9868c56aea77e",
1635
+ "promptSha256": "e996790789038c53f8aece1cea769a864a30699aa034d1774e5ba5a98ec42923",
1636
+ "deterministic": [],
1637
+ "judge": {
1638
+ "verdict": "pass",
1639
+ "reason": "Criterion 1: concrete code `mocker.patch(\"mypkg.mod.requests.get\", return_value=FakeResponse(...))` patches the external HTTP call at point of use. Criterion 2: \"The internal helper stays real\" and \"Zero mocks on the internal helper\"; test comment \"helper ran for real\". Criterion 3: \"Mocking it away means the test only verifies that your mocks agree with each other.\" No fail criterion holds."
1640
+ },
1641
+ "passed": true
1642
+ },
1643
+ {
1644
+ "output": "Mock the external HTTP API call; do not mock the internal helper in the same package.\n\n**Why:**\n\n- **External HTTP API** \u2014 this is a genuine external dependency (network, another service, non-deterministic, slow, rate-limited). Mock it so the test is fast, hermetic, and doesn't hit the network. Patch at the point of use, i.e. where the module-under-test *calls* it:\n ```python\n mocker.patch(\"mypkg.mod.requests\") # or httpx / the client wrapper\n # or, if the module does `from . import client` and calls `client.get(...)`:\n mocker.patch(\"mypkg.mod.client.get\", return_value=fake_response)\n ```\n Patch the *call site* (`mypkg.mod...`), never the definition site (`requests.get` globally, or the library's own module) \u2014 patching the definition site can miss the binding the code actually invokes.\n\n- **Internal helper in the same package** \u2014 this is the real integration surface under test. Mocking it away means the test only checks that the mocked helper agrees with the mocked API, and it stops verifying that the helper is called correctly and that its real output is wired through. Let it run for real. If the helper itself is slow or needs its own external calls mocked, that mock belongs on *its* external dependency, not on the helper.\n\n**Rule of thumb:** mock across a process/network/other-service boundary; don't mock across a package boundary within the code you own.\n\nIf the internal helper is expensive or needs setup (e.g. a DB), that's a case for a fixture that provisions a real or in-memory dependency for the helper, not for mocking the helper out of existence.\n\n```python\ndef test_process_calls_api(monkeypatch):\n # mock the external boundary only\n monkeypatch.setattr(\"mypkg.mod.http_client.fetch\", lambda url: {\"ok\": True})\n result = mypkg.mod.process(\"input\") # internal helper runs for real\n assert result.status == \"ok\"\n```",
1645
+ "outputSha256": "b7942085cbab541d270941b8176bff927bfcc3ef6e16be2172a46a0bbddf9f84",
1646
+ "promptSha256": "e996790789038c53f8aece1cea769a864a30699aa034d1774e5ba5a98ec42923",
1647
+ "deterministic": [],
1648
+ "judge": {
1649
+ "verdict": "pass",
1650
+ "reason": "Concrete HTTP mock shown: mocker.patch(\"mypkg.mod.client.get\", ...) and monkeypatch.setattr(\"mypkg.mod.http_client.fetch\", ...). Explicitly leaves internal helper unmocked (\"Let it run for real\", comment \"internal helper runs for real\"). Explains why: mocking helper means test only checks mocked helper agrees with mocked API, stops verifying real wiring."
1651
+ },
1652
+ "passed": true
1653
+ },
1654
+ {
1655
+ "output": "Mock the external HTTP API only. Do not mock the internal helper.\n\nWhy:\n\n- **Internal helper (same package): call it for real.** The rule in this skill is \"mock external dependencies, not internal ones.\" A test where an internal collaborator is mocked only verifies that the mocks agree with each other \u2014 it stops the test from exercising real integration between the function under test and its own module's logic. If the helper makes the test slow or awkward, that is usually a signal to test the helper separately (directly) and let this test call through, rather than to stub it out.\n- **External HTTP API: mock it.** Network calls are external, non-deterministic, slow, and often unavailable in CI. They belong behind a patch.\n- **Patch at the point of use, not the definition site.** If the function does `import requests` and calls `requests.get(...)`, patch the symbol where the module under test looks it up \u2014 e.g. `mocker.patch(\"mypkg.mod.requests\")` or `mocker.patch(\"mypkg.mod.requests.get\")`. If the module under test did `from httpx import Client` and calls `Client(...)`, patch `mypkg.mod.Client`, not `httpx.Client`. Patching the definition site (`requests.get`) is the common mistake: it can silently fail to intercept if the module already bound the name, and it leaks into other tests.\n\nSketch:\n\n```python\ndef test_fetch_thing_calls_api(mocker):\n mock_get = mocker.patch(\"mypkg.mod.requests.get\")\n mock_get.return_value.json.return_value = {\"id\": 1}\n mock_get.return_value.raise_for_status.return_value = None\n\n result = mypkg.mod.fetch_thing(1)\n\n mock_get.assert_called_once_with(\"https://api.example.com/things/1\", timeout=...)\n assert result.id == 1\n```\n\nNotes that matter for this specific shape:\n\n- Assert the **API contract** at the boundary: the URL, method, params, and how the response is translated into the return value. That is the part the internal helper cannot tell you.\n- Because the helper runs for real, keep the test input one that drives the helper down a representative path; the parametrize cases (happy path, empty/None, `pytest.raises(SpecificError)`) go on **this** test over the same mocked API.\n- Mock only the one external seam. If the helper itself calls a second external service (e.g. a DB), that is a second external dependency and a second mock \u2014 still nothing internal.\n- Never edit the source under test to make this pass; if it fails, fix the test (max 3 iterations, then report the failing case).",
1656
+ "outputSha256": "353ca366c11d83ee1720a20a84fef6bb0a7fdf80bfd88464988c8ba4c432c160",
1657
+ "promptSha256": "e996790789038c53f8aece1cea769a864a30699aa034d1774e5ba5a98ec42923",
1658
+ "deterministic": [],
1659
+ "judge": {
1660
+ "verdict": "pass",
1661
+ "reason": "Criterion 1: concrete mock shown, `mocker.patch(\"mypkg.mod.requests.get\")` in the sketch, patched at point of use. Criterion 2: explicitly says \"Do not mock the internal helper\" and \"call it for real\"; no patch/spy on the helper. Criterion 3: explains mocking internal collaborator \"only verifies that the mocks agree with each other\" and stops real integration. No fail criterion holds."
1662
+ },
1663
+ "passed": true
1664
+ },
1665
+ {
1666
+ "output": "Mock the external HTTP API only \u2014 not the internal helper.\n\nThe rule from the skill: mock at the point of use, and mock external dependencies, never internal collaborators in the same package. My rationale: \"Mock external dependencies (network, filesystem, other services), not internal modules under the same package,\" and \"I'll mock the internal helper so the test is simpler\" is explicitly listed as a red flag, because \"a test whose internal collaborators are all mocked only checks that the mocks agree with each other.\"\n\nThe un-mocked internal helper earns its keep: the test then verifies the mocked HTTP response was translated correctly into whatever the helper returns/produces, which is the behavior the function under test is actually responsible for.\n\nIn pytest terms:\n\n```python\n# tests/test_client.py\nimport pytest\n\n\ndef test_fetch_user_returns_parsed_profile(mocker):\n # Patch the HTTP call where the module under test uses it\n # (not where requests is defined).\n mock_get = mocker.patch(\"mypkg.client.requests.get\")\n\n mock_response = mocker.Mock()\n mock_response.status_code = 200\n mock_response.json.return_value = {\"id\": 1, \"name\": \"Ada\"}\n mock_get.return_value = mock_response\n\n from mypkg.client import fetch_user\n\n result = fetch_user(1)\n\n # The internal helper's real behavior is exercised here;\n # only the network boundary is faked.\n assert result.name == \"Ada\"\n mock_get.assert_called_once()\n```\n\nKey points:\n\n- Patch path is `\"mypkg.client.requests.get\"` \u2014 the module that *uses* `requests`, not `requests.get` itself. Patching at the definition site (`mocker.patch(\"requests.get\")`) can be bypassed by import-time bindings and is the classic source of \"the mock didn't take\" confusion.\n- The internal helper is called for real. If it has its own edge cases (empty input, missing keys), cover those in separate tests of that helper directly rather than stubbing it out from the caller's test.\n- If the helper itself makes additional external calls, those get mocked too \u2014 mock external boundaries regardless of which layer reaches them, but never mock the package's own functions.\n\nIf a project-wide `conftest.py` (or an existing autouse fixture like a \"no network\" guard) already stubs the HTTP client, use that instead of a local `mocker.patch` \u2014 the skill's Step 1\u20132 says to match the project's existing mock conventions rather than introducing a second style.",
1667
+ "outputSha256": "7d2275bfdafef6f1658b0a0306f7af0f54336c70d5d05ac4f311b359e72d28bf",
1668
+ "promptSha256": "e996790789038c53f8aece1cea769a864a30699aa034d1774e5ba5a98ec42923",
1669
+ "deterministic": [],
1670
+ "judge": {
1671
+ "verdict": "pass",
1672
+ "reason": "Criterion 1: concrete code `mocker.patch(\"mypkg.client.requests.get\")` patches the external HTTP call at point of use. Criterion 2: states internal helper is called for real, not stubbed. Criterion 3: explains mocked internal collaborators 'only checks that the mocks agree with each other.' No fail criterion: helper is not patched/spied."
1673
+ },
1674
+ "passed": true
1675
+ },
1676
+ {
1677
+ "output": "Mock the external HTTP call; do not mock the internal helper.\n\nThe external HTTP API is the classic external dependency \u2014 it crosses a process/network boundary, is slow, non-deterministic, and unavailable in CI. Patch it at the point of use in the module under test, e.g. `mocker.patch(\"mypkg.mod.requests.get\")` (or `httpx.Client.post`, the `requests` session method, etc. \u2014 patch wherever the module actually looks up the reference, not the library's definition site).\n\nThe internal helper in the same package should be exercised for real. Mocking it would only verify that your mock agrees with your own code: the test would pass even if the helper were broken or if the function stopped calling it correctly. Per this skill's rules, mocks are for external dependencies (network, filesystem, other services), not internal modules in the same package.\n\nConcretely, for something like:\n\n```python\n# mypkg/mod.py\nimport requests\nfrom .helpers import normalize # internal helper\n\ndef fetch_and_normalize(url: str) -> dict:\n raw = requests.get(url, timeout=5).json() # external\n return normalize(raw) # internal \u2014 leave real\n```\n\nwrite:\n\n```python\n# tests/test_mod.py\nimport pytest\nfrom mypkg.mod import fetch_and_normalize\n\n@pytest.mark.parametrize(\n \"raw, expected\",\n [\n ({\"a\": 1}, \"a=1\"),\n ({}, \"\"),\n ],\n ids=[\"single-key\", \"empty\"],\n)\ndef test_fetch_and_normalize(mocker, raw, expected):\n mock_get = mocker.patch(\"mypkg.mod.requests.get\")\n mock_get.return_value.json.return_value = raw\n\n assert fetch_and_normalize(\"http://example.test/x\") == expected\n mock_get.assert_called_once_with(\"http://example.test/x\", timeout=5)\n```\n\nNotes:\n- Patch target is `mypkg.mod.requests.get` \u2014 the name as `mypkg.mod` resolves it at call time \u2014 not `requests.get` globally.\n- `normalize` runs for real, so the test actually exercises the integration between the two calls.\n- If `normalize` itself talks to something external (a DB, another service), mock *that* dependency \u2014 still not `normalize` itself.\n- Only if `normalize` were genuinely unreachable-to-test (e.g. a slow side-effecting collaborator not owned by this package) would you consider mocking it, and then it belongs behind a fixture, not a scattered patch.",
1678
+ "outputSha256": "f461979efc3ae532d1e4fdc09296094691bb38554d08817e0f3a4630f1f805bc",
1679
+ "promptSha256": "e996790789038c53f8aece1cea769a864a30699aa034d1774e5ba5a98ec42923",
1680
+ "deterministic": [],
1681
+ "judge": {
1682
+ "verdict": "pass",
1683
+ "reason": "Criterion 1: concrete patch shown, `mocker.patch(\"mypkg.mod.requests.get\")` with return_value.json set. Criterion 2: explicitly leaves `normalize` real \u2014 code comment \"internal \u2014 leave real\" and note \"`normalize` runs for real\". Criterion 3: explains mocking the helper \"would only verify that your mock agrees with your own code\" and test would pass even if helper broken. No fail criterion: helper is not patched/spied."
1684
+ },
1685
+ "passed": true
1686
+ },
1687
+ {
1688
+ "output": "Only the external HTTP API call should be mocked. The internal helper in the same package should NOT be mocked \u2014 it runs for real, because mocking it would only verify that a fake agrees with the test, not that the two functions actually work together. Per the skill: \"Mock external dependencies (network, filesystem, other services), not internal modules under the same package.\"\n\nConcretely, the mock belongs at the point of use, not the definition site. So if the API client is imported into the module under test like `from mypkg.http import fetch` or `import requests`, patch the name as it appears in the module under test:\n\n```python\n# tests/test_service.py\nimport pytest\nfrom mypkg.service import process # the function under test\n\n\ndef test_process_returns_parsed_payload(mocker):\n # Patch at the point of use: the attribute the module under test looks up.\n mock_get = mocker.patch(\"mypkg.service.requests.get\")\n\n mock_response = mock_get.return_value\n mock_response.status_code = 200\n mock_response.json.return_value = {\"items\": [1, 2, 3]}\n\n result = process(\"query\")\n\n # Assert the HTTP boundary was hit with the expected call, not how.\n mock_get.assert_called_once_with(\"https://api.example.com/query\")\n assert result == ... # whatever the real internal helper + parsing produces\n\n\ndef test_process_raises_on_api_error(mocker):\n mock_get = mocker.patch(\"mypkg.service.requests.get\")\n mock_get.return_value.status_code = 500\n mock_get.return_value.raise_for_status.side_effect = httpx.HTTPStatusError(...)\n\n with pytest.raises(...):\n process(\"query\")\n```\n\nKey points:\n\n- **Do not** write `mocker.patch(\"mypkg.helpers.internal_helper\")`. That internal helper is part of the behavior under test \u2014 leave it real so the test actually exercises the integration between the helper and the request/parse logic.\n- **Do not** mock at the definition site (e.g. `mocker.patch(\"requests.get\")` globally, or patching the source module the HTTP client lives in). Patch where the module under test resolves the name, which is `mypkg.service.requests.get` if it's imported as `import requests` inside that module.\n- If the internal helper itself performs I/O (reads a file, hits another service), mock *that dependency* at the helper's point of use \u2014 still not the helper's own logic. The rule is \"mock external dependencies,\" not \"mock internal modules.\"\n- Keep the HTTP surface minimal: fake the response object (status code, `.json()`, `.raise_for_status()`) rather than standing up a real server. If the project already uses `responses` or `respx` for this, match that convention instead of `mocker.patch`.\n\nFix the test, not the source, if something fails \u2014 and only touch test files and `conftest.py`.",
1689
+ "outputSha256": "9d083e6a1b5d4f51cc9116feb60c6e91c4ea5ccbee8ea969e217ada583d2b463",
1690
+ "promptSha256": "e996790789038c53f8aece1cea769a864a30699aa034d1774e5ba5a98ec42923",
1691
+ "deterministic": [],
1692
+ "judge": {
1693
+ "verdict": "pass",
1694
+ "reason": "Criterion 1: concrete code `mocker.patch(\"mypkg.service.requests.get\")` patches the external HTTP call at point of use. Criterion 2: explicitly says internal helper should NOT be mocked, leave it real, and warns against `mocker.patch(\"mypkg.helpers.internal_helper\")`. Criterion 3: explains mocking the helper would only verify a fake agrees with the test, not that the functions work together. No fail criterion holds."
1695
+ },
1696
+ "passed": true
1697
+ },
1698
+ {
1699
+ "output": "Mock only the external HTTP API call, not the internal helper.\n\nPatch where the HTTP client is used, at the point of use:\n\n```python\ndef test_fetch_user(mocker):\n mock_get = mocker.patch(\"mypkg.mod.requests.get\")\n mock_get.return_value.json.return_value = {\"id\": 1, \"name\": \"Ada\"}\n mock_get.return_value.raise_for_status.return_value = None\n\n result = fetch_user(1)\n\n assert result.name == \"Ada\"\n mock_get.assert_called_once()\n```\n\nWhy not the internal helper: the helper is part of the same package, so mocking it makes the test assert that your mocks agree with each other rather than that the real logic works. Exercising the real helper is exactly the integration the test should verify \u2014 an internal collaborator mocked away stops the test checking anything meaningful.\n\nIf you must patch an internal helper because it is genuinely slow or non-deterministic (e.g. a clock, RNG, or filesystem walk inside it), patch the narrowest seam \u2014 the clock/RNG module \u2014 not the helper function itself. That keeps the helper's own logic under test while removing the nondeterminism.\n\nTwo mechanical notes:\n\n- Patch the name the module under test looks up (`mypkg.mod.requests.get` or a `mypkg.mod.http_client`), never the definition site (`requests.get`), because the lookup happens in the importing module's namespace.\n- Assert on the observable result and one call-level check (`assert_called_once_with` or equivalent) rather than asserting the mock returned what you told it to return \u2014 the latter is a tautology.\n\nIf the file already has an HTTP fixture or a recorded-response helper in Step 1 discovery, use that instead of an ad-hoc `mocker.patch` so the new test matches project convention.",
1700
+ "outputSha256": "f13f4ce3883560e368fea8932a92f6d0e08d662d7115c3611d241c8e1bb29e98",
1701
+ "promptSha256": "e996790789038c53f8aece1cea769a864a30699aa034d1774e5ba5a98ec42923",
1702
+ "deterministic": [],
1703
+ "judge": {
1704
+ "verdict": "pass",
1705
+ "reason": "Answer concretely patches external HTTP at point of use: `mocker.patch(\"mypkg.mod.requests.get\")` with return_value setup. It explicitly leaves the internal helper unmocked, stating the helper is part of the same package and exercising it is the integration to verify. It explains mocking the helper would only assert mocks agree with each other, not real logic. No internal helper is patched."
1706
+ },
1707
+ "passed": true
1708
+ },
1709
+ {
1710
+ "output": "Mock the external HTTP API call, not the internal helper.\n\nConcretely, patch the HTTP call at the point where your module uses it, e.g.:\n\n```python\n# mypkg/service.py\nimport requests\n\ndef get_user(user_id):\n data = _normalize(user_id) # internal helper, same package\n resp = requests.get(f\"https://api.example.com/users/{user_id}\")\n return resp.json()\n```\n\n```python\n# tests/test_service.py\ndef test_get_user(mocker):\n mock_get = mocker.patch(\"mypkg.service.requests.get\")\n mock_get.return_value.json.return_value = {\"id\": 1, \"name\": \"Ada\"}\n\n result = get_user(1)\n\n assert result == {\"id\": 1, \"name\": \"Ada\"}\n mock_get.assert_called_once_with(\"https://api.example.com/users/1\")\n```\n\nWhy this split:\n\n- **External HTTP API \u2192 mock.** It is outside your process, non-deterministic, slow, and may have side effects. Patch it at the point of use (`mypkg.service.requests.get`), not at the definition site (`requests.get`), so the patch targets the name your module actually resolves. This is also where a red-flag applies: mocking an internal collaborator makes the test only check that your mocks agree with each other, not that the code integrates.\n- **Internal helper \u2192 do not mock.** It lives in the same package and is part of what you are testing. Mocking it means the test would pass even if the helper were broken, and it couples the test to an implementation detail you are free to refactor. Let it run for real so the test exercises actual integration.\n\nTwo practical notes:\n\n1. If the HTTP call is made via `requests`, patching `requests.get`/`requests.post` at the use site is usually enough. If the project has a configured HTTP client or a session object, patch that object's method instead \u2014 match the convention you find in Step 1 of the workflow (read a neighboring test file to see how the project already mocks HTTP).\n2. For async code that awaits an HTTP client, use `pytest.mark.asyncio` and patch the awaited call (e.g. an `httpx.AsyncClient.get` or the project's client method); do not write a sync test that never awaits the coroutine.\n\nIf you also want to isolate the helper's behavior, test the helper directly in its own test with its own inputs rather than replacing it in this test. Keep this test focused on one behavior: that the function sends the right request and returns the parsed result.",
1711
+ "outputSha256": "ab6a71c5464645f9d7f58ef8c7dad9eef40f9dc6bdec4a6816ddd28bc213e4fe",
1712
+ "promptSha256": "e996790789038c53f8aece1cea769a864a30699aa034d1774e5ba5a98ec42923",
1713
+ "deterministic": [],
1714
+ "judge": {
1715
+ "verdict": "pass",
1716
+ "reason": "Criterion 1: concrete code `mocker.patch(\"mypkg.service.requests.get\")` patches the external HTTP call at point of use. Criterion 2: explicitly says internal helper `_normalize` is not mocked, \"Let it run for real so the test exercises actual integration.\" Criterion 3: explains mocking internal collaborator \"makes the test only check that your mocks agree with each other, not that the code integrates.\" No fail criterion holds."
1717
+ },
1718
+ "passed": true
1719
+ },
1720
+ {
1721
+ "output": "Mock the external HTTP API; leave the internal helper real.\n\nThe rule from Step 3/Step 4 is to patch at the point of use and only external dependencies. The internal helper in the same package is a collaborator you want the test to exercise for real \u2014 if you mock it away, the test only verifies that your mocks agree with each other and stops checking real integration, which is exactly the \"mock the internal helper so the test is simpler\" red flag.\n\nSo:\n\n- External HTTP call: mock it, patched where the module under test looks it up, e.g. `mocker.patch(\"mypkg.mod.requests.get\")` (or whatever the module actually references \u2014 the name as used at the call site, not the definition site in the HTTP library).\n- Internal helper in the same package: do not patch. Let it run. Give it inputs that make its real behavior contribute to the assertion, and if it in turn calls the external API, that call is already covered by the patch above.\n\nIf the internal helper has its own side effects or external dependencies that are awkward in a unit test, that is a sign to test the helper separately (its own test with its own targeted mock), not to mock the helper inside this test.\n\nSketch:\n\n```python\ndef test_fetch_and_transform(mocker):\n mock_get = mocker.patch(\"mypkg.mod.requests.get\")\n mock_get.return_value.json.return_value = {\"value\": 21}\n\n result = fetch_and_transform(\"thing\")\n\n # internal helper ran for real: assert on its transformation,\n # not on a stubbed return value\n assert result == 42\n mock_get.assert_called_once_with(\"https://api.example.com/thing\")\n```\n\nKeep the mock at the boundary, let the internal helper do its real work, and assert on the combined result.",
1722
+ "outputSha256": "fbd06d85e3055e11d188e82afb52f05a68078aac7100040eeb23ab62ac6a48eb",
1723
+ "promptSha256": "e996790789038c53f8aece1cea769a864a30699aa034d1774e5ba5a98ec42923",
1724
+ "deterministic": [],
1725
+ "judge": {
1726
+ "verdict": "pass",
1727
+ "reason": "Criterion 1: concrete mock at point of use, `mocker.patch(\"mypkg.mod.requests.get\")` with call-site note. Criterion 2: explicitly says internal helper 'do not patch. Let it run' and comment 'internal helper ran for real'. Criterion 3: explains mocking it 'only verifies that your mocks agree with each other and stops checking real integration'. No fail criterion: helper not patched/spied."
1728
+ },
1729
+ "passed": true
1730
+ },
1731
+ {
1732
+ "output": "Mock the external HTTP API call; do not mock the internal helper in the same package.\n\nWhy:\n\n- **External HTTP API \u2192 mock it.** It's a real external dependency with network I/O, latency, and non-determinism, so the test would be slow and flaky without patching it. Use `mocker.patch` (pytest-mock) or `unittest.mock.patch` at the point where your module looks it up, e.g. if the module does `import requests` and calls `requests.get(...)`, patch `\"mypkg.mod.requests.get\"`; if it does `from httpx import Client` and uses `Client`, patch the name as bound in the module under test (`\"mypkg.mod.Client\"`). Patch at the *use site*, not the definition site.\n\n- **Internal helper in the same package \u2192 do not mock.** The whole point of the test is to verify the function's real behavior end-to-end through its own code path. If you mock the internal helper, you're only asserting that your mocks agree with each other, not that the function actually works. Exercise the real helper.\n\n```python\n# tests/test_mod.py\nimport pytest\nfrom mypkg import mod\n\n\ndef test_fetch_and_process(mocker):\n # Mock only the external HTTP API, at the use site in this module.\n fake_response = mocker.Mock()\n fake_response.json.return_value = {\"value\": 21}\n fake_response.raise_for_status.return_value = None\n mocker.patch(\"mypkg.mod.requests.get\", return_value=fake_response)\n\n # The internal helper runs for real.\n result = mod.fetch_and_process(\"https://example.test/data\")\n\n assert result == 42 # whatever the real helper computes from {\"value\": 21}\n```\n\nA few refinements worth applying:\n\n- If the internal helper is genuinely expensive or has its own side effects, prefer injecting a seam (pass a dependency in) over mocking it \u2014 but unless it's a true external boundary, still don't patch it out.\n- Mock at the boundaries of your system (network, filesystem, clock, third-party services), not internal collaborators.\n- Assert on the real behavior produced through the real helper \u2014 that's what gives the test value.\n\nAfter writing it, run the test and confirm it passes without touching any source file under `mypkg`.",
1733
+ "outputSha256": "78a35ee9c5f4abee99dde399b272415804460bd1b04d6187fbfb7308f7f9e347",
1734
+ "promptSha256": "e996790789038c53f8aece1cea769a864a30699aa034d1774e5ba5a98ec42923",
1735
+ "deterministic": [],
1736
+ "judge": {
1737
+ "verdict": "pass",
1738
+ "reason": "Criterion 1: concrete code `mocker.patch(\"mypkg.mod.requests.get\", return_value=fake_response)` patches the external HTTP call at the use site. Criterion 2: explicitly says internal helper runs for real, comment '# The internal helper runs for real.' and no patch on it. Criterion 3: explains mocking the helper would only assert mocks agree with each other, not that the function works."
1739
+ },
1740
+ "passed": true
1741
+ }
1742
+ ]
1743
+ }
1744
+ ],
1745
+ "verdict": "pass",
1746
+ "scope": "bundled",
1747
+ "skillDigest": "3b9d03e1a2a7bfae96eb7e234daf0521ab85c086086e2776c6124ed74732b6b3",
1748
+ "catalogDigest": "4f4016d410837e394a27e5b247e38ef2f57a1ee0baba4436ce7d3d71e223333d",
1749
+ "judgePromptVersion": "2026-09-25.1",
1750
+ "runner": "deepseek",
1751
+ "model": "deepseek-chat",
1752
+ "runnerPromptVersion": "2026-09-25.1",
1753
+ "recordedAt": "2026-09-25T05:17:31.953Z",
1754
+ "judge": "deepseek",
1755
+ "judgeModel": "deepseek-chat"
1756
+ }
1757
+ ]
1758
+ }