evalrx 0.1.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (339) hide show
  1. evalrx/__init__.py +139 -0
  2. evalrx/agent_assets/__init__.py +2 -0
  3. evalrx/agent_assets/skills/README.md +28 -0
  4. evalrx/agent_assets/skills/eval-chart-style/SKILL.md +172 -0
  5. evalrx/agent_assets/skills/evalrx-report-ui/SKILL.md +116 -0
  6. evalrx/agent_assets/skills/nature-figure/LICENSE +201 -0
  7. evalrx/agent_assets/skills/nature-figure/README.md +412 -0
  8. evalrx/agent_assets/skills/nature-figure/SKILL.md +60 -0
  9. evalrx/agent_assets/skills/nature-figure/manifest.yaml +59 -0
  10. evalrx/agent_assets/skills/nature-figure/references/api.md +436 -0
  11. evalrx/agent_assets/skills/nature-figure/references/backend-selection.md +100 -0
  12. evalrx/agent_assets/skills/nature-figure/references/chart-types.md +281 -0
  13. evalrx/agent_assets/skills/nature-figure/references/common-patterns.md +350 -0
  14. evalrx/agent_assets/skills/nature-figure/references/demos.md +65 -0
  15. evalrx/agent_assets/skills/nature-figure/references/design-theory.md +439 -0
  16. evalrx/agent_assets/skills/nature-figure/references/figure-contract.md +93 -0
  17. evalrx/agent_assets/skills/nature-figure/references/figure-legend-conventions.md +71 -0
  18. evalrx/agent_assets/skills/nature-figure/references/nature-2026-observations.md +112 -0
  19. evalrx/agent_assets/skills/nature-figure/references/qa-contract.md +119 -0
  20. evalrx/agent_assets/skills/nature-figure/references/r-template-index.md +66 -0
  21. evalrx/agent_assets/skills/nature-figure/references/r-workflow.md +161 -0
  22. evalrx/agent_assets/skills/nature-figure/references/tutorials.md +251 -0
  23. evalrx/agent_assets/skills/nature-figure/static/core/contract.md +29 -0
  24. evalrx/agent_assets/skills/nature-figure/static/core/stance.md +37 -0
  25. evalrx/agent_assets/skills/nature-figure/static/fragments/backend/python.md +37 -0
  26. evalrx/agent_assets/skills/nature-figure/static/fragments/backend/r.md +44 -0
  27. evalrx/agent_assets/skills/outcome-driver-analysis/SKILL.md +213 -0
  28. evalrx/agent_assets/skills/outcome-driver-analysis/assets/analysis_report_template.md +53 -0
  29. evalrx/agent_assets/skills/outcome-driver-analysis/references/model_selection.md +72 -0
  30. evalrx/agent_assets/skills/outcome-driver-analysis/scripts/explanatory_var_eda.R +130 -0
  31. evalrx/agent_assets/skills/outcome-driver-analysis/scripts/explanatory_var_eda.py +150 -0
  32. evalrx/agent_assets/skills/outcome-driver-analysis/scripts/fit_outcome_model.R +181 -0
  33. evalrx/agent_assets/skills/outcome-driver-analysis/scripts/fit_outcome_model.py +186 -0
  34. evalrx/agent_assets/skills/outcome-driver-analysis/scripts/univariate_eda.R +149 -0
  35. evalrx/agent_assets/skills/outcome-driver-analysis/scripts/univariate_eda.py +177 -0
  36. evalrx/agent_assets/skills.py +27 -0
  37. evalrx/agent_runtime/__init__.py +78 -0
  38. evalrx/agent_runtime/_docker_runner.py +89 -0
  39. evalrx/agent_runtime/cli_runtime.py +103 -0
  40. evalrx/agent_runtime/cli_transcript.py +138 -0
  41. evalrx/agent_runtime/cli_types.py +68 -0
  42. evalrx/agent_runtime/codegen/__init__.py +5 -0
  43. evalrx/agent_runtime/codegen/runner.py +94 -0
  44. evalrx/agent_runtime/experiment_harness.py +117 -0
  45. evalrx/agent_runtime/factory.py +102 -0
  46. evalrx/agent_runtime/json_shape.py +44 -0
  47. evalrx/agent_runtime/judges/__init__.py +28 -0
  48. evalrx/agent_runtime/judges/agy.py +179 -0
  49. evalrx/agent_runtime/judges/autodetect.py +135 -0
  50. evalrx/agent_runtime/judges/claude.py +159 -0
  51. evalrx/agent_runtime/judges/codex.py +120 -0
  52. evalrx/agent_runtime/providers/__init__.py +21 -0
  53. evalrx/agent_runtime/providers/antigravity.py +31 -0
  54. evalrx/agent_runtime/providers/base.py +145 -0
  55. evalrx/agent_runtime/providers/claude_code.py +49 -0
  56. evalrx/agent_runtime/providers/codex.py +37 -0
  57. evalrx/agent_runtime/providers/gemini_cli.py +26 -0
  58. evalrx/agent_runtime/providers/kimi_cli.py +27 -0
  59. evalrx/agent_runtime/providers/opencode.py +27 -0
  60. evalrx/agent_runtime/providers/registry.py +58 -0
  61. evalrx/agent_runtime/sandbox.py +517 -0
  62. evalrx/agent_runtime/skill_audit.py +143 -0
  63. evalrx/agent_runtime/skills/__init__.py +19 -0
  64. evalrx/agent_runtime/skills/installer.py +68 -0
  65. evalrx/agent_runtime/skills/prompt_policy.py +86 -0
  66. evalrx/agent_runtime/skills/resolver.py +19 -0
  67. evalrx/analysis/__init__.py +132 -0
  68. evalrx/analysis/adjudicate.py +154 -0
  69. evalrx/analysis/analysis_module.py +361 -0
  70. evalrx/analysis/api.py +171 -0
  71. evalrx/analysis/case_studio.py +651 -0
  72. evalrx/analysis/cli.py +114 -0
  73. evalrx/analysis/dashboard.py +350 -0
  74. evalrx/analysis/eval_case_matrix.py +118 -0
  75. evalrx/analysis/eval_viz_theme.py +833 -0
  76. evalrx/analysis/explore_run.py +333 -0
  77. evalrx/analysis/explorer.py +1276 -0
  78. evalrx/analysis/failure_modes.py +607 -0
  79. evalrx/analysis/fused_pipeline.py +489 -0
  80. evalrx/analysis/holdout.py +300 -0
  81. evalrx/analysis/hypothesis_agent.py +230 -0
  82. evalrx/analysis/narration.py +177 -0
  83. evalrx/analysis/operationalize.py +442 -0
  84. evalrx/analysis/plain_language.py +42 -0
  85. evalrx/analysis/planner.py +283 -0
  86. evalrx/analysis/probe_search.py +203 -0
  87. evalrx/analysis/profile.py +268 -0
  88. evalrx/analysis/prompts/__init__.py +0 -0
  89. evalrx/analysis/prompts/explorer.py +417 -0
  90. evalrx/analysis/prompts/failure_modes.py +33 -0
  91. evalrx/analysis/prompts/holdout.py +27 -0
  92. evalrx/analysis/prompts/hypothesis_agent.py +78 -0
  93. evalrx/analysis/prompts/run_codebase.py +47 -0
  94. evalrx/analysis/prompts/stats_agent.py +72 -0
  95. evalrx/analysis/prompts/stats_tool_generator.py +43 -0
  96. evalrx/analysis/result_marker.py +47 -0
  97. evalrx/analysis/run_codebase.py +242 -0
  98. evalrx/analysis/run_view.py +205 -0
  99. evalrx/analysis/stage_views.py +93 -0
  100. evalrx/analysis/stats_agent.py +944 -0
  101. evalrx/analysis/stats_tool_agent.py +261 -0
  102. evalrx/analysis/stats_tool_generator.py +415 -0
  103. evalrx/analysis/stats_tools.py +1153 -0
  104. evalrx/analysis/trajectory_records.py +193 -0
  105. evalrx/analysis/workbench.py +431 -0
  106. evalrx/analyzers/__init__.py +42 -0
  107. evalrx/analyzers/agent/__init__.py +25 -0
  108. evalrx/analyzers/agent/counterfactual.py +84 -0
  109. evalrx/analyzers/agent/first_error_judge.py +96 -0
  110. evalrx/analyzers/agent/ignored_obs.py +81 -0
  111. evalrx/analyzers/agent/loop_detect.py +79 -0
  112. evalrx/analyzers/agent/reliability.py +165 -0
  113. evalrx/analyzers/agent/tool_shap.py +225 -0
  114. evalrx/analyzers/agent/trajectory_rubric.py +168 -0
  115. evalrx/analyzers/attention/__init__.py +19 -0
  116. evalrx/analyzers/attention/relative_attn.py +610 -0
  117. evalrx/analyzers/attention/rollout.py +73 -0
  118. evalrx/analyzers/attention/sink.py +56 -0
  119. evalrx/analyzers/attention/summary.py +190 -0
  120. evalrx/analyzers/attribution/__init__.py +6 -0
  121. evalrx/analyzers/attribution/generic_attn.py +31 -0
  122. evalrx/analyzers/attribution/gradcam.py +30 -0
  123. evalrx/analyzers/base.py +12 -0
  124. evalrx/analyzers/geometry/__init__.py +6 -0
  125. evalrx/analyzers/geometry/cka.py +70 -0
  126. evalrx/analyzers/geometry/linear_probe.py +157 -0
  127. evalrx/analyzers/hallucination/__init__.py +9 -0
  128. evalrx/analyzers/hallucination/chair.py +78 -0
  129. evalrx/analyzers/hallucination/opera.py +29 -0
  130. evalrx/analyzers/hallucination/pope.py +119 -0
  131. evalrx/analyzers/hallucination/selfcheck.py +155 -0
  132. evalrx/analyzers/hallucination/vcd.py +29 -0
  133. evalrx/analyzers/lens/__init__.py +7 -0
  134. evalrx/analyzers/lens/layer_contrast.py +133 -0
  135. evalrx/analyzers/lens/logit_lens.py +138 -0
  136. evalrx/analyzers/lens/tuned_lens.py +30 -0
  137. evalrx/analyzers/patching/__init__.py +5 -0
  138. evalrx/analyzers/patching/causal_trace.py +30 -0
  139. evalrx/analyzers/perturbation/__init__.py +23 -0
  140. evalrx/analyzers/perturbation/_shapley.py +54 -0
  141. evalrx/analyzers/perturbation/context_shap.py +174 -0
  142. evalrx/analyzers/perturbation/cot_faithfulness.py +239 -0
  143. evalrx/analyzers/perturbation/format_sensitivity.py +237 -0
  144. evalrx/analyzers/perturbation/mm_shap.py +146 -0
  145. evalrx/analyzers/perturbation/modality_ablation.py +196 -0
  146. evalrx/analyzers/perturbation/perturbation_battery.py +274 -0
  147. evalrx/analyzers/perturbation/prompt_contrast.py +265 -0
  148. evalrx/analyzers/perturbation/rise.py +94 -0
  149. evalrx/analyzers/perturbation/vl_shap.py +102 -0
  150. evalrx/analyzers/reasoning/__init__.py +33 -0
  151. evalrx/analyzers/reasoning/_text.py +328 -0
  152. evalrx/analyzers/reasoning/answer_extraction_audit.py +327 -0
  153. evalrx/analyzers/reasoning/arith_audit.py +226 -0
  154. evalrx/analyzers/reasoning/contamination.py +214 -0
  155. evalrx/analyzers/reasoning/knowledge_split.py +253 -0
  156. evalrx/analyzers/reasoning/self_repair.py +246 -0
  157. evalrx/analyzers/reasoning/step_rollout_value.py +216 -0
  158. evalrx/analyzers/reasoning/termination_audit.py +258 -0
  159. evalrx/analyzers/uncertainty/__init__.py +18 -0
  160. evalrx/analyzers/uncertainty/calibration.py +174 -0
  161. evalrx/analyzers/uncertainty/coverage_gap.py +199 -0
  162. evalrx/analyzers/uncertainty/entropy.py +90 -0
  163. evalrx/analyzers/uncertainty/logprob_entropy.py +69 -0
  164. evalrx/analyzers/uncertainty/self_consistency.py +204 -0
  165. evalrx/analyzers/uncertainty/verbalized_conf.py +64 -0
  166. evalrx/cli.py +411 -0
  167. evalrx/config.py +77 -0
  168. evalrx/contract/__init__.py +179 -0
  169. evalrx/contract/common.py +452 -0
  170. evalrx/contract/emit.py +948 -0
  171. evalrx/contract/export.py +237 -0
  172. evalrx/contract/m1.py +325 -0
  173. evalrx/contract/m2.py +317 -0
  174. evalrx/contract/m3.py +165 -0
  175. evalrx/contract/m4.py +130 -0
  176. evalrx/contract/m5.py +292 -0
  177. evalrx/contract/methodology.py +76 -0
  178. evalrx/contract/pre_m1.py +58 -0
  179. evalrx/contract/typescript.py +140 -0
  180. evalrx/core/__init__.py +85 -0
  181. evalrx/core/analyzer.py +174 -0
  182. evalrx/core/capability.py +54 -0
  183. evalrx/core/case.py +443 -0
  184. evalrx/core/experiment.py +106 -0
  185. evalrx/core/model.py +198 -0
  186. evalrx/core/pipeline.py +42 -0
  187. evalrx/core/registry.py +142 -0
  188. evalrx/core/result.py +64 -0
  189. evalrx/core/spec.py +173 -0
  190. evalrx/core/tokentype.py +165 -0
  191. evalrx/core/tool.py +92 -0
  192. evalrx/datasets/__init__.py +41 -0
  193. evalrx/datasets/base.py +68 -0
  194. evalrx/datasets/gui_os.py +52 -0
  195. evalrx/datasets/llm_qa.py +57 -0
  196. evalrx/datasets/pure_qa.py +12 -0
  197. evalrx/datasets/vlm_qa.py +695 -0
  198. evalrx/datasets/web_search_qa.py +52 -0
  199. evalrx/eval_agent/__init__.py +341 -0
  200. evalrx/eval_agent/_tools.py +81 -0
  201. evalrx/eval_agent/ab_runner.py +50 -0
  202. evalrx/eval_agent/agentic/__init__.py +43 -0
  203. evalrx/eval_agent/agentic/actions.py +216 -0
  204. evalrx/eval_agent/agentic/board.py +107 -0
  205. evalrx/eval_agent/agentic/loop.py +190 -0
  206. evalrx/eval_agent/agentic/tools.py +538 -0
  207. evalrx/eval_agent/checkpoint.py +57 -0
  208. evalrx/eval_agent/cli_agent.py +59 -0
  209. evalrx/eval_agent/cli_skills.py +5 -0
  210. evalrx/eval_agent/evolution.py +396 -0
  211. evalrx/eval_agent/git_manager.py +215 -0
  212. evalrx/eval_agent/hypothesis.py +172 -0
  213. evalrx/eval_agent/label_quarantine.py +209 -0
  214. evalrx/eval_agent/legacy.py +530 -0
  215. evalrx/eval_agent/log_schema.py +497 -0
  216. evalrx/eval_agent/loop.py +2159 -0
  217. evalrx/eval_agent/loop_reports.py +116 -0
  218. evalrx/eval_agent/model_instrumentation.py +282 -0
  219. evalrx/eval_agent/narration.py +193 -0
  220. evalrx/eval_agent/nl_runner.py +460 -0
  221. evalrx/eval_agent/orchestrator.py +61 -0
  222. evalrx/eval_agent/preregister.py +93 -0
  223. evalrx/eval_agent/prompts/__init__.py +1 -0
  224. evalrx/eval_agent/prompts/agentic.py +46 -0
  225. evalrx/eval_agent/prompts/case_discovery.py +25 -0
  226. evalrx/eval_agent/prompts/diagnosis.py +125 -0
  227. evalrx/eval_agent/prompts/experiment_writer.py +265 -0
  228. evalrx/eval_agent/prompts/explore_step.py +37 -0
  229. evalrx/eval_agent/prompts/fix_agent.py +257 -0
  230. evalrx/eval_agent/prompts/hypothesis_tester.py +15 -0
  231. evalrx/eval_agent/prompts/nl_runner.py +38 -0
  232. evalrx/eval_agent/prompts/probe_agent.py +25 -0
  233. evalrx/eval_agent/prompts/probe_candidate_generator.py +14 -0
  234. evalrx/eval_agent/prompts/probe_generator.py +35 -0
  235. evalrx/eval_agent/prompts/whitebox_probe_generator.py +38 -0
  236. evalrx/eval_agent/report.py +58 -0
  237. evalrx/eval_agent/run_context.py +354 -0
  238. evalrx/eval_agent/run_log.schema.json +1215 -0
  239. evalrx/eval_agent/run_logger_v2.py +1764 -0
  240. evalrx/eval_agent/run_metadata.py +208 -0
  241. evalrx/eval_agent/stages/__init__.py +56 -0
  242. evalrx/eval_agent/stages/case_discovery.py +293 -0
  243. evalrx/eval_agent/stages/diagnosis.py +1017 -0
  244. evalrx/eval_agent/stages/experiment_writer.py +1634 -0
  245. evalrx/eval_agent/stages/fix_agent.py +3916 -0
  246. evalrx/eval_agent/stages/fix_internals.py +499 -0
  247. evalrx/eval_agent/stages/fix_pipeline.py +725 -0
  248. evalrx/eval_agent/stages/fix_tiers.py +187 -0
  249. evalrx/eval_agent/stages/fix_tools.py +1034 -0
  250. evalrx/eval_agent/stages/hypothesis_tester.py +1014 -0
  251. evalrx/eval_agent/stages/probe.py +439 -0
  252. evalrx/eval_agent/stages/probe_agent.py +1079 -0
  253. evalrx/eval_agent/stages/probe_candidate_generator.py +128 -0
  254. evalrx/eval_agent/stages/probe_generator.py +326 -0
  255. evalrx/eval_agent/stages/probe_search_agent.py +106 -0
  256. evalrx/eval_agent/stages/protocol.py +112 -0
  257. evalrx/eval_agent/stages/repair_catalog.py +273 -0
  258. evalrx/eval_agent/stages/surgery.py +524 -0
  259. evalrx/eval_agent/stages/whitebox_probe_generator.py +351 -0
  260. evalrx/eval_agent/store.py +231 -0
  261. evalrx/logging_utils.py +112 -0
  262. evalrx/models/__init__.py +161 -0
  263. evalrx/models/_discover.py +101 -0
  264. evalrx/models/agent.py +380 -0
  265. evalrx/models/backends/__init__.py +58 -0
  266. evalrx/models/backends/api.py +169 -0
  267. evalrx/models/backends/base.py +57 -0
  268. evalrx/models/backends/gemini_compat.py +579 -0
  269. evalrx/models/backends/hf_local.py +2074 -0
  270. evalrx/models/backends/openai_compat.py +301 -0
  271. evalrx/models/backends/vllm_offline.py +116 -0
  272. evalrx/models/base.py +24 -0
  273. evalrx/models/blackbox/__init__.py +4 -0
  274. evalrx/models/blackbox/agent.py +31 -0
  275. evalrx/models/blackbox/base.py +29 -0
  276. evalrx/models/blackbox/gemini.py +279 -0
  277. evalrx/models/blackbox/llm_api.py +17 -0
  278. evalrx/models/blackbox/vlm_api.py +17 -0
  279. evalrx/models/compose.py +66 -0
  280. evalrx/models/inference.py +88 -0
  281. evalrx/models/paper_methods/__init__.py +8 -0
  282. evalrx/models/paper_methods/aad.py +53 -0
  283. evalrx/models/paper_methods/ifcd.py +204 -0
  284. evalrx/models/paper_methods/pai.py +164 -0
  285. evalrx/models/paper_methods/tcd.py +202 -0
  286. evalrx/models/paper_methods/vcd.py +45 -0
  287. evalrx/models/paper_methods/vicrop.py +137 -0
  288. evalrx/models/toolcodec.py +143 -0
  289. evalrx/models/tools/__init__.py +20 -0
  290. evalrx/models/tools/perception.py +300 -0
  291. evalrx/models/tools/visual.py +174 -0
  292. evalrx/models/whitebox/__init__.py +26 -0
  293. evalrx/models/whitebox/agent.py +31 -0
  294. evalrx/models/whitebox/base.py +24 -0
  295. evalrx/models/whitebox/qwen.py +61 -0
  296. evalrx/models/whitebox/qwen2_5_omni.py +29 -0
  297. evalrx/models/whitebox/qwen2_audio.py +25 -0
  298. evalrx/models/whitebox/qwen_omni.py +53 -0
  299. evalrx/models/whitebox/qwen_vl.py +62 -0
  300. evalrx/observability/__init__.py +21 -0
  301. evalrx/observability/envelope.py +122 -0
  302. evalrx/observability/outbox.py +111 -0
  303. evalrx/observability/tracer.py +882 -0
  304. evalrx/reporting/__init__.py +28 -0
  305. evalrx/reporting/case_study.py +947 -0
  306. evalrx/reporting/compiler.py +587 -0
  307. evalrx/reporting/dynamic.py +1882 -0
  308. evalrx/reporting/html_report.py +2225 -0
  309. evalrx/reporting/langfuse_exporter.py +38 -0
  310. evalrx/reporting/langfuse_source.py +155 -0
  311. evalrx/reporting/model.py +151 -0
  312. evalrx/reporting/run_events.py +184 -0
  313. evalrx/reporting/server.py +557 -0
  314. evalrx/reporting/stages.py +58 -0
  315. evalrx/reporting/static_export.py +142 -0
  316. evalrx/reporting/web_dist/index.html +146 -0
  317. evalrx/specs.py +727 -0
  318. evalrx/stats/__init__.py +47 -0
  319. evalrx/stats/api.py +192 -0
  320. evalrx/stats/bootstrap.py +86 -0
  321. evalrx/stats/ebh.py +27 -0
  322. evalrx/stats/evalue.py +98 -0
  323. evalrx/stats/friedman.py +138 -0
  324. evalrx/stats/mcnemar.py +40 -0
  325. evalrx/stats/multiplicity.py +159 -0
  326. evalrx/stats/subset_sampling.py +55 -0
  327. evalrx/term_links.py +43 -0
  328. evalrx/viz/__init__.py +7 -0
  329. evalrx/viz/labels.py +77 -0
  330. evalrx/viz/prompts.py +39 -0
  331. evalrx/viz/renderer.py +590 -0
  332. evalrx/viz/schema.py +36 -0
  333. evalrx/viz/style.py +134 -0
  334. evalrx-0.1.2.dist-info/METADATA +532 -0
  335. evalrx-0.1.2.dist-info/RECORD +339 -0
  336. evalrx-0.1.2.dist-info/WHEEL +5 -0
  337. evalrx-0.1.2.dist-info/entry_points.txt +3 -0
  338. evalrx-0.1.2.dist-info/licenses/LICENSE +121 -0
  339. evalrx-0.1.2.dist-info/top_level.txt +1 -0
@@ -0,0 +1,251 @@
1
+ # Tutorials — Nature Figure Making
2
+
3
+ End-to-end walkthroughs for the most common publication figure types.
4
+ All examples use helpers from [api.md](api.md) and patterns from [common-patterns.md](common-patterns.md).
5
+ For real production scripts and output previews from figures4papers, open [demos.md](demos.md).
6
+
7
+ ---
8
+
9
+ ## Tutorial 1: Grouped bar chart (multi-metric comparison)
10
+
11
+ **Goal**: Several methods compared across multiple metrics. Legend in a dedicated panel.
12
+ When methods belong to related families, use one coherent baseline family plus one coherent hero family.
13
+
14
+ ```python
15
+ import os
16
+ import numpy as np
17
+ import matplotlib.pyplot as plt
18
+ from matplotlib import gridspec
19
+
20
+ # --- Style ---
21
+ plt.rcParams['font.family'] = 'sans-serif'
22
+ plt.rcParams['font.sans-serif'] = ['Arial']
23
+ plt.rcParams['svg.fonttype'] = 'none'
24
+ plt.rcParams['font.size'] = 24
25
+ plt.rcParams['axes.spines.right'] = False
26
+ plt.rcParams['axes.spines.top'] = False
27
+ plt.rcParams['axes.linewidth'] = 3
28
+
29
+ # --- Data ---
30
+ methods = ['ResNet1d18', 'ResNet1d34', 'ECGFounder', 'CSFM-Tiny', 'CSFM-Base', 'CSFM-Large']
31
+ colors = ['#484878', '#7884B4', '#B4C0E4', '#E4E4F0', '#E4CCD8', '#F0C0CC']
32
+ metrics = ['Metric 1', 'Metric 2', 'Metric 3']
33
+ mean = {
34
+ 'Metric 1': np.array([0.81, 0.83, 0.86, 0.89, 0.91, 0.92]),
35
+ 'Metric 2': np.array([0.63, 0.67, 0.71, 0.74, 0.77, 0.79]),
36
+ 'Metric 3': np.array([0.41, 0.45, 0.49, 0.53, 0.56, 0.58]),
37
+ }
38
+ std = {k: v * 0.03 for k, v in mean.items()} # placeholder
39
+
40
+ # --- Figure ---
41
+ fig = plt.figure(figsize=(28, 6))
42
+ gs = gridspec.GridSpec(1, len(metrics) + 1) # +1 for legend panel
43
+
44
+ handles, labels = None, None
45
+ for col, metric in enumerate(metrics):
46
+ ax = fig.add_subplot(gs[col])
47
+ bars = ax.bar(
48
+ range(len(methods)),
49
+ mean[metric],
50
+ yerr=std[metric],
51
+ capsize=5,
52
+ color=colors,
53
+ label=methods,
54
+ error_kw={'elinewidth': 2, 'capthick': 2},
55
+ )
56
+ if col == 0:
57
+ handles, labels = ax.get_legend_handles_labels()
58
+ ax.set_xticks([])
59
+ y_vals = mean[metric]
60
+ margin = (y_vals.max() - y_vals.min()) * 0.15
61
+ ax.set_ylim([y_vals.min() - margin, y_vals.max() + margin])
62
+ ax.set_ylabel(metric, fontsize=32)
63
+
64
+ # Legend-only panel
65
+ ax_leg = fig.add_subplot(gs[-1])
66
+ ax_leg.legend(handles, labels, fontsize=28, loc='center', frameon=False)
67
+ ax_leg.set_axis_off()
68
+
69
+ fig.tight_layout(pad=2)
70
+ os.makedirs('./figures', exist_ok=True)
71
+ fig.savefig('./figures/comparison.png', dpi=300)
72
+ fig.savefig('./figures/comparison.pdf', dpi=300)
73
+ plt.close(fig)
74
+ ```
75
+
76
+ ---
77
+
78
+ ## Tutorial 2: Ablation bar chart (alpha-graduated, horizontal)
79
+
80
+ **Goal**: Same method with components progressively added; alpha encodes completeness.
81
+
82
+ ```python
83
+ import os
84
+ import numpy as np
85
+ import matplotlib.pyplot as plt
86
+
87
+ plt.rcParams['font.family'] = 'sans-serif'
88
+ plt.rcParams['font.sans-serif'] = ['Arial']
89
+ plt.rcParams['svg.fonttype'] = 'none'
90
+ plt.rcParams['font.size'] = 24
91
+ plt.rcParams['axes.spines.right'] = False
92
+ plt.rcParams['axes.spines.top'] = False
93
+ plt.rcParams['axes.linewidth'] = 3
94
+
95
+ configs = ['None', '+ Module A', '+ Module B', '+ Module C', 'Full']
96
+ values = np.array([0.72, 0.78, 0.81, 0.84, 0.88])
97
+ stds = np.array([0.02, 0.02, 0.01, 0.01, 0.01])
98
+
99
+ n = len(configs)
100
+ blue_rgb = (0.215686, 0.458824, 0.729412) # #3775BA
101
+ alphas = np.linspace(0.2, 1.0, n)
102
+ colors = [(blue_rgb[0], blue_rgb[1], blue_rgb[2], a) for a in alphas]
103
+
104
+ fig, ax = plt.subplots(figsize=(12, 6))
105
+ ax.barh(range(n), values, xerr=stds,
106
+ color=colors, ecolor='k', capsize=5)
107
+ ax.set_yticks(range(n))
108
+ ax.set_yticklabels(configs)
109
+ ax.set_xlim([values.min() - 0.05, values.max() + 0.03])
110
+ ax.set_xlabel('Score', fontsize=32)
111
+
112
+ fig.tight_layout(pad=2)
113
+ os.makedirs('./figures', exist_ok=True)
114
+ fig.savefig('./figures/ablation.png', dpi=300)
115
+ plt.close(fig)
116
+ ```
117
+
118
+ ---
119
+
120
+ ## Tutorial 3: Multi-panel trend with shared legend
121
+
122
+ **Goal**: Two trend panels (e.g., train/val curves) and a legend-only third panel.
123
+
124
+ ```python
125
+ import os
126
+ import numpy as np
127
+ import matplotlib.pyplot as plt
128
+
129
+ plt.rcParams['font.family'] = 'sans-serif'
130
+ plt.rcParams['font.sans-serif'] = ['Arial']
131
+ plt.rcParams['svg.fonttype'] = 'none'
132
+ plt.rcParams['font.size'] = 15
133
+ plt.rcParams['axes.spines.right'] = False
134
+ plt.rcParams['axes.spines.top'] = False
135
+ plt.rcParams['axes.linewidth'] = 2
136
+
137
+ methods = ['Baseline', 'CSFM-Tiny', 'CSFM-Base', 'CSFM-Large']
138
+ colors = ['#7884B4', '#E4E4F0', '#E4CCD8', '#F0C0CC']
139
+ x = np.arange(0, 100, 5)
140
+
141
+ fig, axes = plt.subplots(1, 3, figsize=(18, 5))
142
+
143
+ for panel_idx, (ax, panel_name) in enumerate(zip(axes[:2], ['Training', 'Validation'])):
144
+ for method, color in zip(methods, colors):
145
+ y = 0.48 + 0.42 * (1 - np.exp(-x / 30)) + np.random.randn(len(x)) * 0.01
146
+ if method == 'Baseline':
147
+ y -= 0.03
148
+ elif method == 'CSFM-Tiny':
149
+ y += 0.00
150
+ elif method == 'CSFM-Base':
151
+ y += 0.02
152
+ elif method == 'CSFM-Large':
153
+ y += 0.03
154
+ ax.plot(x, y, color=color, lw=2.5, marker='o', markersize=6, label=method)
155
+ ax.set_title(panel_name, fontsize=18)
156
+ ax.set_xlabel('Epoch', fontsize=16)
157
+ ax.set_ylabel('Loss', fontsize=16)
158
+ if panel_idx == 0:
159
+ handles, labels = ax.get_legend_handles_labels()
160
+
161
+ # Legend-only panel
162
+ axes[2].legend(handles, labels, fontsize=14, loc='center', frameon=False)
163
+ axes[2].set_axis_off()
164
+
165
+ fig.tight_layout(pad=2)
166
+ os.makedirs('./figures', exist_ok=True)
167
+ fig.savefig('./figures/trends.png', dpi=300)
168
+ fig.savefig('./figures/trends.pdf', dpi=300)
169
+ plt.close(fig)
170
+ ```
171
+
172
+ ---
173
+
174
+ ## Tutorial 4: Heatmap with dual colormaps (positive/negative columns)
175
+
176
+ **Goal**: Score matrix where positive = Reds, negative = Blues_r. Cell text auto-contrasted.
177
+
178
+ ```python
179
+ import os
180
+ import numpy as np
181
+ import matplotlib as mpl
182
+ import matplotlib.pyplot as plt
183
+
184
+ plt.rcParams['font.family'] = 'sans-serif'
185
+ plt.rcParams['font.sans-serif'] = ['Arial']
186
+ plt.rcParams['svg.fonttype'] = 'none'
187
+ plt.rcParams['font.size'] = 16
188
+ plt.rcParams['axes.spines.right'] = False
189
+ plt.rcParams['axes.spines.top'] = False
190
+ plt.rcParams['axes.linewidth'] = 2
191
+
192
+ # matrix: rows = methods, cols = metrics (alternating positive/negative directions)
193
+ methods = ['Method A', 'Method B', 'Method C', 'Method D']
194
+ metrics = ['Score (+)', 'Error (-)', 'F1 (+)', 'Loss (-)']
195
+ matrix = np.array([
196
+ [0.88, 0.12, 0.85, 0.20],
197
+ [0.81, 0.18, 0.78, 0.28],
198
+ [0.75, 0.25, 0.72, 0.35],
199
+ [0.70, 0.30, 0.68, 0.40],
200
+ ])
201
+
202
+ fig, ax = plt.subplots(figsize=(10, 6))
203
+ n_rows, n_cols = matrix.shape
204
+ vmin, vmax = matrix.min(0), matrix.max(0)
205
+
206
+ for j in range(n_cols):
207
+ is_positive = (j % 2 == 0)
208
+ cmap = plt.cm.Reds if is_positive else plt.cm.Blues_r
209
+ cmap = cmap.copy()
210
+ norm = mpl.colors.Normalize(
211
+ vmin=0 if is_positive else vmax[j],
212
+ vmax=vmax[j] if is_positive else 0
213
+ )
214
+ ax.imshow(matrix[:, j:j+1], cmap=cmap, norm=norm,
215
+ aspect='auto', extent=[j-0.5, j+0.5, 0, n_rows], origin='lower')
216
+
217
+ for (i, j), val in np.ndenumerate(matrix):
218
+ is_positive = (j % 2 == 0)
219
+ cmap = plt.cm.Reds if is_positive else plt.cm.Blues_r
220
+ norm = mpl.colors.Normalize(vmin=0 if is_positive else vmax[j],
221
+ vmax=vmax[j] if is_positive else 0)
222
+ r, g, b, _ = cmap(norm(val))
223
+ lum = 0.299*r + 0.587*g + 0.114*b
224
+ color = 'white' if lum < 0.5 else 'black'
225
+ ax.text(j, i + 0.5, f'{val:.2f}', ha='center', va='center',
226
+ fontsize=13, color=color)
227
+
228
+ ax.set_xlim(-0.5, n_cols - 0.5)
229
+ ax.set_xticks(np.arange(n_cols))
230
+ ax.set_xticklabels(metrics, rotation=30, ha='right', fontsize=14)
231
+ ax.tick_params(axis='x', bottom=False, top=False, length=0)
232
+ ax.set_yticks(np.arange(n_rows) + 0.5)
233
+ ax.set_yticklabels(methods, fontsize=14)
234
+ ax.set_frame_on(False)
235
+ ax.invert_yaxis()
236
+
237
+ fig.tight_layout(pad=2)
238
+ os.makedirs('./figures', exist_ok=True)
239
+ fig.savefig('./figures/heatmap.png', dpi=300)
240
+ plt.close(fig)
241
+ ```
242
+
243
+ ---
244
+
245
+ ## Related files
246
+
247
+ - [SKILL.md](../SKILL.md) — When to use this skill
248
+ - [api.md](api.md) — Reusable helper implementations
249
+ - [common-patterns.md](common-patterns.md) — Layout and encoding patterns used above
250
+ - [design-theory.md](design-theory.md) — Why these choices exist
251
+ - [chart-types.md](chart-types.md) — Radar, 3D sphere, scatter, fill_between
@@ -0,0 +1,29 @@
1
+ # Figure contract before plotting
2
+
3
+ A publication-quality scientific figure is a visual argument, not an isolated pretty plot. Every figure starts from a claim, an evidence hierarchy, and a review-risk check before code or aesthetics. Before generating or editing code, establish the contract below.
4
+
5
+ ## Backend selection is a blocking gate
6
+
7
+ If the user has not explicitly chosen Python or R in the current request or provided a clearly language-specific input file/workflow, ask one concise question: **Python or R?** Then stop and wait for the user's answer. Do not generate mock data, write scripts, create figures, or choose Python/R by default. This overrides general autonomy/default-execution behavior for figure tasks.
8
+
9
+ Only recommend a backend when the user explicitly asks you to choose or recommend one. In that case, use `references/backend-selection.md`, state the reason, and then proceed with the recommended backend.
10
+
11
+ ## The selected backend is exclusive
12
+
13
+ Once Python or R is selected, every plotting script, preview image, SVG/PDF/TIFF/PNG export, QA render, and visual workaround must be produced by that same backend. Do not use Python to draw a preview for an R figure, and do not use R to draw a preview for a Python figure, even if the selected runtime or packages are missing locally. The non-selected language may only be used for non-visual file inspection or data conversion when it does not open a graphics device, import plotting libraries, create image/vector files, or change the final visual appearance.
14
+
15
+ ## Missing runtime/package rule
16
+
17
+ After the backend is selected, check the selected runtime early (`Rscript`/R for R; Python and required plotting packages for Python). If the selected runtime or required packages are unavailable, stop before rendering and report the exact blocker. You may provide a selected-backend script and installation commands, or ask permission to install dependencies, but you must not fall back to the other language to make a substitute figure.
18
+
19
+ ## The five-point contract
20
+
21
+ 1. **Core conclusion**: write the one-sentence claim the figure must defend.
22
+ 2. **Evidence chain**: map each planned panel to the claim, and drop panels that do not carry a unique piece of evidence.
23
+ 3. **Archetype**: classify the figure as `quantitative grid`, `schematic-led composite`, `image plate + quant`, or `asymmetric mixed-modality figure`.
24
+ 4. **Backend**: use the selected Python or R track exclusively for all figure drawing, previewing, exporting, and visual QA. Do not cross-render with the other language.
25
+ 5. **Journal/export contract**: set final dimensions, editable text, source data, statistics, image-integrity notes, and export formats before styling.
26
+
27
+ The highest-priority rule is: **the chart serves the scientific logic**. Aesthetic polish, template matching, and complex layout are subordinate to making the core conclusion clear, defensible, and reviewable.
28
+
29
+ For the full method to convert a request into core conclusion, evidence hierarchy, panel map, and review-risk checks, open `references/figure-contract.md`.
@@ -0,0 +1,37 @@
1
+ # Default operating stance
2
+
3
+ The older Python/matplotlib rules in this skill remain valid. The skill also supports R, especially `ggplot2 + patchwork + ComplexHeatmap + ggrepel + svglite/cairo_pdf + ragg`.
4
+
5
+ ## Color policy
6
+
7
+ Prefer **unified method families across all panels** over maximal hue separation. For dense Nature Machine Intelligence-style figure pages, use the low-saturation `NMI pastel` family described in `references/api.md` and reserve green/red mainly for gains, drops, and other directional cues.
8
+
9
+ ## Stance
10
+
11
+ - Start by classifying the requested figure into one of four archetypes: `quantitative grid`, `schematic-led composite`, `image plate + quant`, or `asymmetric mixed-modality figure`.
12
+ - Prefer one **hero panel** plus subordinate evidence panels over filling the canvas with equal-sized subplots.
13
+ - If the user asks for a single chart, still identify its role in the manuscript claim: discovery, mechanism, validation, comparison, robustness, or clinical/biological relevance.
14
+ - Keep the background white for plots and diagrams; switch to black only for microscopy / volume-rendering image plates.
15
+ - Prefer direct labels over legends when categories are spatially fixed or the legend would force unnecessary eye travel.
16
+ - Keep one restrained palette per figure: usually one neutral family, one signal family, and one accent family.
17
+ - Treat statistics, `n`, error-bar definitions, source-data traceability, and image-integrity notes as part of the figure, not as optional caption cleanup.
18
+ - When the user asks for broad `Nature` style rather than ML/NMI-specific style, read `references/nature-2026-observations.md` before choosing layout.
19
+ - When the user references `figures4papers` or the older `scientific-figure-making` skill, treat this skill as the successor and open `references/demos.md` for bundled Python demo scripts.
20
+
21
+ ## User-facing privacy rule
22
+
23
+ Do not disclose private local paths, private filenames, chat-attachment names, internal reference filenames, template identifiers, or the provenance of private working materials in user-facing replies, generated code comments, figure legends, reports, or manuscript text. Use generic descriptions such as "the provided R template collection", "a private working draft", or "the internal figure contract". If the user provides a private plotting template collection, use it only as an internal adaptation source and do not reveal its path, filenames, or provenance. Only reveal an exact path or source file when the user explicitly asks for that audit trail.
24
+
25
+ ## When to load this skill
26
+
27
+ - Python or R figures for **papers, slides, or reports** targeting Nature, Science, Cell, NeurIPS, ICLR, or similar venues.
28
+ - Requests involving **grouped bars, trend lines, heatmaps, radar plots, multi-panel grids**, or **PDF/SVG/high-DPI** output.
29
+ - Any mention of "Nature style", "publication figure", "paper figure", "SCI figure", "figures4papers", "scientific-figure-making", "R plotting template", or "high-quality scientific plot".
30
+ - Requests to improve a figure's logic, aesthetics, panel layout, figure legend, export quality, or journal-readiness.
31
+
32
+ ## When NOT to load
33
+
34
+ - Plotly, Altair, Bokeh, or other interactive/web-first plotting.
35
+ - EDA-only plots without a publication target.
36
+ - Primary workflow is 3D, GIS, or non-scientific illustration tooling.
37
+ - Illustrator / Figma–first layout.
@@ -0,0 +1,37 @@
1
+ # Backend: Python (matplotlib / seaborn)
2
+
3
+ **Python-only execution rule.** When the user has selected Python, do all figure drawing, previewing, exporting, and visual QA in Python. Do not call R/ggplot2, ComplexHeatmap, patchwork, or any R graphics device to create a temporary preview, fallback export, or layout approximation. If Python or required Python plotting packages are missing, stop before rendering and report the missing dependency. You may still write the Python script, provide `pip`/environment install commands, or ask permission to install dependencies, but do not cross-render the figure in R.
4
+
5
+ ## Python quick-start
6
+
7
+ ```python
8
+ import matplotlib as mpl
9
+ import matplotlib.pyplot as plt
10
+
11
+ mpl.rcParams.update({
12
+ "font.family": "sans-serif",
13
+ "font.sans-serif": ["Arial", "Helvetica", "DejaVu Sans", "sans-serif"],
14
+ "svg.fonttype": "none", # editable text in SVG
15
+ "pdf.fonttype": 42, # editable TrueType text in PDF
16
+ "font.size": 7, # use 15-24 only for large slide-sized panels
17
+ "axes.spines.right": False,
18
+ "axes.spines.top": False,
19
+ "axes.linewidth": 0.8,
20
+ "legend.frameon": False,
21
+ })
22
+
23
+ def save_pub_py(fig, filename, dpi=600):
24
+ fig.savefig(f"{filename}.svg", bbox_inches="tight")
25
+ fig.savefig(f"{filename}.pdf", bbox_inches="tight")
26
+ fig.savefig(f"{filename}.tiff", dpi=dpi, bbox_inches="tight")
27
+ ```
28
+
29
+ Use `text.usetex = True` only when LaTeX is installed and math-rich labels are required.
30
+
31
+ ## Going deeper
32
+
33
+ - `references/api.md` — Python PALETTE, helper function signatures, validation rules.
34
+ - `references/common-patterns.md` — hero panels, legend-only axes, dark image plates, asymmetric layouts.
35
+ - `references/chart-types.md` — radar, 3D sphere, fill_between, scatter patterns.
36
+ - `references/tutorials.md` — end-to-end walkthroughs for bars, trends, heatmaps.
37
+ - `references/demos.md` — bundled figures4papers Python scripts and output previews.
@@ -0,0 +1,44 @@
1
+ # Backend: R (ggplot2 / patchwork / ComplexHeatmap)
2
+
3
+ **R-only execution rule.** When the user has selected R, do all figure drawing, previewing, exporting, and visual QA in R. Do not call matplotlib/seaborn or any Python graphics device to create a temporary preview, fallback export, or layout approximation. If `Rscript`/R or required R packages are missing, stop before rendering and report the exact blocker. You may still write the R script, provide install commands (for example `install.packages(...)`), or ask permission to install dependencies, but do not cross-render the figure in Python.
4
+
5
+ ## R quick-start
6
+
7
+ ```r
8
+ library(ggplot2)
9
+ library(patchwork)
10
+
11
+ theme_set(
12
+ theme_classic(base_size = 6.5, base_family = "Arial") +
13
+ theme(
14
+ axis.line = element_line(linewidth = 0.35, colour = "black"),
15
+ axis.ticks = element_line(linewidth = 0.35, colour = "black"),
16
+ legend.title = element_text(size = 6.2),
17
+ legend.text = element_text(size = 5.8),
18
+ strip.text = element_text(size = 6.2, face = "bold"),
19
+ plot.title = element_text(size = 7, face = "bold"),
20
+ panel.grid = element_blank()
21
+ )
22
+ )
23
+
24
+ save_pub_r <- function(plot, filename, width_mm = 183, height_mm = 120, dpi = 600) {
25
+ w <- width_mm / 25.4
26
+ h <- height_mm / 25.4
27
+ svglite::svglite(paste0(filename, ".svg"), width = w, height = h)
28
+ print(plot)
29
+ dev.off()
30
+ grDevices::cairo_pdf(paste0(filename, ".pdf"), width = w, height = h, family = "Arial")
31
+ print(plot)
32
+ dev.off()
33
+ ragg::agg_tiff(paste0(filename, ".tiff"), width = w, height = h, units = "in", res = dpi)
34
+ print(plot)
35
+ dev.off()
36
+ }
37
+ ```
38
+
39
+ ## Going deeper
40
+
41
+ - `references/r-workflow.md` — the R plotting workflow when the user provides R scripts, templates, or data.
42
+ - `references/r-template-index.md` — adapt a user-provided or private R template collection without exposing source paths.
43
+ - `references/design-theory.md` — typography, color theory, layout rationale, export policy (backend-agnostic).
44
+ - `references/nature-2026-observations.md` — real Nature page archetypes to match before choosing layout.
@@ -0,0 +1,213 @@
1
+ ---
2
+ name: outcome-driver-analysis
3
+ description: >
4
+ Run a full, disciplined statistical analysis to find what differentiates a binary
5
+ outcome (success vs. fail, pass vs. fail, correct vs. incorrect) using a set of
6
+ explanatory variables the user specifies. Covers exploring the explanatory variables
7
+ themselves (distributions, outliers, missingness, correlations), exploring each
8
+ variable against the outcome (contingency tables for categorical variables,
9
+ distribution comparisons for continuous variables, conditioning on other variables),
10
+ marginal screening, fitting a justified regression model (logistic vs. linear vs.
11
+ mixed-effects, with explicit reasoning), goodness-of-fit diagnostics, result
12
+ visualization, and a plain-language written report. Use this whenever the user has
13
+ examples labeled by a binary outcome plus candidate explanatory variables and wants
14
+ to know what drives the difference -- e.g. analyzing model error cases ("why do
15
+ these examples fail"), pass/fail experiment results, or any dataset framed as
16
+ success/failure with covariates -- even if they don't use the word "statistics."
17
+ Do NOT use this for pure simulation studies with no real data, generating
18
+ paper-ready LaTeX tables for an already-written paper, or benchmarking many methods
19
+ against each other across datasets -- those are out of scope.
20
+ version: 0.2.0
21
+ license: MIT
22
+ compatibility: Claude Code project-scoped skill. Assumes R and/or Python available on PATH; pick whichever is appropriate per task rather than requiring both.
23
+ metadata:
24
+ tags: statistics research-workflow eda logistic-regression mixed-effects reproducibility R Python
25
+ agentskills_spec: "1.0"
26
+ ---
27
+
28
+ # Outcome driver analysis
29
+
30
+ You are running a statistical analysis to explain a binary outcome (success/fail,
31
+ pass/fail, correct/incorrect) using explanatory variables the user provides. Work
32
+ through the pipeline below in order -- each step's findings feed the next one. Do not
33
+ skip straight to modeling: the exploratory steps surface issues (outliers, missing
34
+ data, collinearity, confounding) that change how the model should be built and read.
35
+
36
+ ## Scope
37
+
38
+ This skill is for analyzing a real dataset with a binary outcome and candidate
39
+ explanatory variables -- most often model error analysis (why did these examples
40
+ fail?), but equally applicable to any pass/fail or success/failure outcome with
41
+ covariates. It is not for simulation studies with no real data, generating
42
+ publication-ready LaTeX tables for a paper that's already written, or benchmarking
43
+ many methods against each other across datasets.
44
+
45
+ ## Chart types and palettes: defer to a chart-style skill when present
46
+
47
+ This skill decides WHAT to visualize at each step (the statistical intent);
48
+ it does not own HOW. When a dedicated chart-style skill (e.g.
49
+ `eval-chart-style`) is installed alongside this one, that skill's chart-type
50
+ policy and palette GOVERN every figure called for below — read the concrete
51
+ figure prescriptions in steps 2, 3 and 7 as the standalone fallback for when
52
+ no chart-style skill is present, never as an override of one.
53
+
54
+ ## Choosing R or Python
55
+
56
+ Pick per task, in this order:
57
+ 1. If the user is continuing an existing project, match whatever language that
58
+ project's code already uses.
59
+ 2. Otherwise, pick based on the modeling need identified in step 5: mixed-effects /
60
+ hierarchical models are frequently more ergonomic in R (`lme4`, `glmmTMB`), while
61
+ plain logistic regression, screening, and diagnostics are equally well supported in
62
+ both R and Python (`statsmodels`, `scikit-learn`).
63
+ 3. If genuinely ambiguous, ask the user.
64
+
65
+ Bundled helper scripts exist in both languages under `scripts/` -- use the one that
66
+ matches your choice; they are templates to adapt to the actual variable names and
67
+ data, not black boxes to run unmodified.
68
+
69
+ ## 1. Intake -- clarify variables and structure
70
+
71
+ Before writing any code, confirm (from what the user already said, or by asking):
72
+ - Which column is the binary outcome, and which columns are the explanatory variables
73
+ to investigate.
74
+ - The research question or problem behind the analysis -- what decision or
75
+ understanding this is meant to support. Record it; the final report must be framed
76
+ around it, not written as a generic statistical summary.
77
+ - Each explanatory variable's type: categorical, continuous, or count.
78
+ - Whether observations are clustered or repeated -- e.g. multiple examples from the
79
+ same underlying model, prompt template, dataset, or subject. This is needed later to
80
+ decide whether a mixed-effects model is warranted.
81
+
82
+ ## 2. Explanatory-variable EDA
83
+
84
+ Before relating anything to the outcome, characterize the explanatory variables on
85
+ their own terms:
86
+ - **Distribution**: histogram (continuous) or bar chart (categorical) per variable.
87
+ - **Outliers**: flag with an IQR or z-score rule for continuous variables; flag rare
88
+ categories for categorical variables.
89
+ - **Missingness**: count and pattern per variable. Spot-check whether missingness
90
+ itself is associated with the outcome (missing values are often not random).
91
+ - **Inter-variable structure**: correlation matrix for continuous-continuous pairs,
92
+ Cramer's V for categorical-categorical pairs, correlation ratio (eta-squared) or
93
+ ANOVA for mixed pairs. This surfaces redundant or collinear explanatory variables
94
+ early, before they cause multicollinearity problems at the modeling stage.
95
+
96
+ Use `scripts/explanatory_var_eda.R` or `.py` as a starting template.
97
+
98
+ ## 3. Per-variable exploration vs. outcome
99
+
100
+ For each explanatory variable, relate it to the outcome:
101
+ - **Categorical variable** -> contingency table (variable x outcome) with row/column
102
+ proportions; chi-square test of independence, or Fisher's exact test when any
103
+ expected cell count is small (below ~5).
104
+ - **Continuous variable** -> a per-group distribution-comparison figure
105
+ (standalone fallback: side-by-side boxplot; an installed chart-style skill's
106
+ distribution-first policy — e.g. violin + jittered points — takes
107
+ precedence), plus a group-comparison test: Welch's t-test if roughly normal,
108
+ Mann-Whitney U as the robust default when normality is doubtful.
109
+ - Always report an effect size next to the test, not just a p-value: Cramer's V or
110
+ odds ratio for categorical variables, Cohen's d or rank-biserial correlation for
111
+ continuous variables.
112
+ - **Conditioning**: repeat the above stratified by, or faceted on, a plausible
113
+ confounding or interacting variable. Flag relationships that appear, vanish, or
114
+ reverse within strata (a Simpson's-paradox pattern) -- these are candidates for an
115
+ interaction term or control variable in the model stage, not things to quietly drop.
116
+
117
+ Use `scripts/univariate_eda.R` or `.py` as a starting template.
118
+
119
+ ## 4. Marginal variable screening
120
+
121
+ Before committing to a multivariable model, screen each explanatory variable for a
122
+ marginal (unadjusted) signal against the outcome: fit a univariate logistic regression
123
+ per variable (or reuse the chi-square/t-test/Mann-Whitney results from step 3), and
124
+ rank by p-value or by AIC improvement over the null model.
125
+
126
+ This matters most when there are many candidate variables -- it narrows the field to a
127
+ manageable candidate set for the full model. Do not silently drop a variable just
128
+ because its bivariate test wasn't significant; a real effect can be masked by a
129
+ confounder and only emerge once other variables are adjusted for in step 5. Screening
130
+ informs the candidate list, it does not replace the full model.
131
+
132
+ Screening metrics and thresholds are summarized in `references/model_selection.md`.
133
+
134
+ ## 5. Model selection and fitting, with explicit justification
135
+
136
+ Consult `references/model_selection.md` for the full decision table. The core logic:
137
+
138
+ - **Binary outcome -> logistic regression, not linear regression.** State explicitly
139
+ why: a linear model can predict outside [0, 1], and its error/variance structure
140
+ doesn't match a 0/1 target (violates linearity and homoscedasticity assumptions that
141
+ linear regression relies on). Continuous outcomes would call for linear regression
142
+ instead, and count outcomes for Poisson or negative binomial -- this decision logic
143
+ generalizes even though the immediate case here is binary.
144
+ - **Plain GLM vs. mixed-effects (GLMM)**: use a random effect for the clustering
145
+ variable identified in step 1 when observations are not independent (repeated
146
+ examples from the same model, prompt, or dataset). Use a plain GLM when observations
147
+ are reasonably independent. State this decision and the reasoning tied to the actual
148
+ data structure -- don't pick silently.
149
+ - **Candidate predictors**: start from the variables that passed step 4 screening, plus
150
+ any interaction flagged by the step 3 conditioning check.
151
+ - **Per-variable significance in the fitted model**: after fitting, test each
152
+ variable's significance with a Wald test or a likelihood-ratio test (comparing the
153
+ model with and without that term). Report this alongside the bivariate/screening
154
+ results from steps 3-4, so the reader can see whether a variable's marginal signal
155
+ holds up after adjusting for the others, or was actually explained by a confounder.
156
+
157
+ Use `scripts/fit_outcome_model.R` or `.py` as a starting template.
158
+
159
+ ## 6. Goodness-of-fit and diagnostics
160
+
161
+ - Hosmer-Lemeshow test (or a suitable alternative when there are many continuous
162
+ predictors, since Hosmer-Lemeshow can be unreliable there).
163
+ - Deviance or Pearson residuals, binned residual plots.
164
+ - VIF for multicollinearity among the final model's predictors.
165
+ - Influence diagnostics (e.g. Cook's-distance analogs for GLMs).
166
+ - ROC curve and AUC for discrimination; a calibration plot for calibration.
167
+ - For GLMMs specifically: also check random-effect variance estimates, intraclass
168
+ correlation (ICC), and convergence warnings.
169
+
170
+ ## 7. Visualize results
171
+
172
+ - Coefficient / odds-ratio forest plot with confidence intervals.
173
+ - Predicted-probability curves for the key continuous predictors (holding other
174
+ variables at a reference value or mean).
175
+ - ROC curve and calibration plot (carried over from step 6, presented as final results
176
+ rather than diagnostics here).
177
+
178
+ ## 8. Conclusion and report
179
+
180
+ State which variables matter, in which direction, with what size and uncertainty, and
181
+ tie the finding back to the steps 2-5 evidence (distribution/outlier caveats, bivariate
182
+ and screening signal, adjusted-model significance) as corroboration or explanation.
183
+
184
+ Frame the conclusion around the specific research question recorded in step 1 -- not
185
+ as a generic statistical summary. Write it in plain language: the audience is
186
+ scientific researchers who understand research methodology but are not necessarily
187
+ statisticians, so translate statistical results into substantive meaning (for example,
188
+ "cases where X exceeded N were far more likely to fail" rather than reporting only an
189
+ odds ratio and a p-value), while still surfacing the effect size, uncertainty, and any
190
+ caveats a careful reader would need (small samples, assumption violations, correlated
191
+ predictors).
192
+
193
+ Figure styling conventions come from the installed chart-style/polish skills
194
+ (e.g. `eval-chart-style` for chart types and palette, `nature-figure` for
195
+ publication polish) -- do not duplicate those conventions here.
196
+
197
+ Save reproducibility information -- environment/package versions and any random seeds
198
+ used -- to `session_info.txt` alongside the analysis.
199
+
200
+ Start each new analysis under `projects/<analysis-name>/`, with the raw dataset copied
201
+ (read-only) into `data/`, outputs written to `output/figures/` and `output/tables/`,
202
+ and the writeup as `report.md` following `assets/analysis_report_template.md`.
203
+
204
+ ## Reference files
205
+
206
+ - `references/model_selection.md` -- decision tables for outcome type x clustering x
207
+ assumption status -> model family, the univariate-test decision table from step 3,
208
+ and the marginal-screening metrics from step 4. Read this when deciding on a test or
209
+ model.
210
+ - `scripts/explanatory_var_eda.R` / `.py` -- step 2 starting template.
211
+ - `scripts/univariate_eda.R` / `.py` -- steps 3-4 starting template.
212
+ - `scripts/fit_outcome_model.R` / `.py` -- steps 5-7 starting template.
213
+ - `assets/analysis_report_template.md` -- report skeleton for step 8.
@@ -0,0 +1,53 @@
1
+ <!--
2
+ Report skeleton for step 8 of the outcome-driver-analysis skill. Copy this into
3
+ projects/<analysis-name>/report.md and fill it in.
4
+
5
+ Figure/table formatting and prose style are covered by the clear-technical-writing
6
+ skill -- apply its conventions when writing the actual content, don't duplicate them
7
+ here.
8
+ -->
9
+
10
+ # [Analysis title]
11
+
12
+ ## Research question
13
+
14
+ What decision or understanding is this analysis meant to support (from intake, step 1)?
15
+
16
+ ## Data
17
+
18
+ Source, size, outcome definition, explanatory variables and their types, any known
19
+ data-quality caveats.
20
+
21
+ ## Explanatory-variable summary
22
+
23
+ Distributions, outliers, missingness, notable correlations among explanatory variables
24
+ (step 2).
25
+
26
+ ## Univariate findings
27
+
28
+ How each explanatory variable relates to the outcome on its own, and what changes when
29
+ conditioning on other variables (step 3). Marginal screening results (step 4).
30
+
31
+ ## Model and justification
32
+
33
+ Which model was fit and why (outcome type, independence/clustering structure, step 5).
34
+ Per-variable significance after adjusting for other predictors.
35
+
36
+ ## Diagnostics
37
+
38
+ Goodness-of-fit, residuals, multicollinearity, discrimination and calibration (step 6).
39
+
40
+ ## Results
41
+
42
+ Coefficient/odds-ratio plot, predicted-probability curves, ROC and calibration plots
43
+ (step 7).
44
+
45
+ ## Conclusions
46
+
47
+ Plain-language answer to the research question above: which variables matter, in what
48
+ direction, how strongly, and with what caveats. Written for a scientific-researcher
49
+ audience, not a statistics audience (step 8).
50
+
51
+ ## Reproducibility
52
+
53
+ Environment/package versions and random seeds used (see `session_info.txt`).