evalrx 0.1.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- evalrx/__init__.py +139 -0
- evalrx/agent_assets/__init__.py +2 -0
- evalrx/agent_assets/skills/README.md +28 -0
- evalrx/agent_assets/skills/eval-chart-style/SKILL.md +172 -0
- evalrx/agent_assets/skills/evalrx-report-ui/SKILL.md +116 -0
- evalrx/agent_assets/skills/nature-figure/LICENSE +201 -0
- evalrx/agent_assets/skills/nature-figure/README.md +412 -0
- evalrx/agent_assets/skills/nature-figure/SKILL.md +60 -0
- evalrx/agent_assets/skills/nature-figure/manifest.yaml +59 -0
- evalrx/agent_assets/skills/nature-figure/references/api.md +436 -0
- evalrx/agent_assets/skills/nature-figure/references/backend-selection.md +100 -0
- evalrx/agent_assets/skills/nature-figure/references/chart-types.md +281 -0
- evalrx/agent_assets/skills/nature-figure/references/common-patterns.md +350 -0
- evalrx/agent_assets/skills/nature-figure/references/demos.md +65 -0
- evalrx/agent_assets/skills/nature-figure/references/design-theory.md +439 -0
- evalrx/agent_assets/skills/nature-figure/references/figure-contract.md +93 -0
- evalrx/agent_assets/skills/nature-figure/references/figure-legend-conventions.md +71 -0
- evalrx/agent_assets/skills/nature-figure/references/nature-2026-observations.md +112 -0
- evalrx/agent_assets/skills/nature-figure/references/qa-contract.md +119 -0
- evalrx/agent_assets/skills/nature-figure/references/r-template-index.md +66 -0
- evalrx/agent_assets/skills/nature-figure/references/r-workflow.md +161 -0
- evalrx/agent_assets/skills/nature-figure/references/tutorials.md +251 -0
- evalrx/agent_assets/skills/nature-figure/static/core/contract.md +29 -0
- evalrx/agent_assets/skills/nature-figure/static/core/stance.md +37 -0
- evalrx/agent_assets/skills/nature-figure/static/fragments/backend/python.md +37 -0
- evalrx/agent_assets/skills/nature-figure/static/fragments/backend/r.md +44 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/SKILL.md +213 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/assets/analysis_report_template.md +53 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/references/model_selection.md +72 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/scripts/explanatory_var_eda.R +130 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/scripts/explanatory_var_eda.py +150 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/scripts/fit_outcome_model.R +181 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/scripts/fit_outcome_model.py +186 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/scripts/univariate_eda.R +149 -0
- evalrx/agent_assets/skills/outcome-driver-analysis/scripts/univariate_eda.py +177 -0
- evalrx/agent_assets/skills.py +27 -0
- evalrx/agent_runtime/__init__.py +78 -0
- evalrx/agent_runtime/_docker_runner.py +89 -0
- evalrx/agent_runtime/cli_runtime.py +103 -0
- evalrx/agent_runtime/cli_transcript.py +138 -0
- evalrx/agent_runtime/cli_types.py +68 -0
- evalrx/agent_runtime/codegen/__init__.py +5 -0
- evalrx/agent_runtime/codegen/runner.py +94 -0
- evalrx/agent_runtime/experiment_harness.py +117 -0
- evalrx/agent_runtime/factory.py +102 -0
- evalrx/agent_runtime/json_shape.py +44 -0
- evalrx/agent_runtime/judges/__init__.py +28 -0
- evalrx/agent_runtime/judges/agy.py +179 -0
- evalrx/agent_runtime/judges/autodetect.py +135 -0
- evalrx/agent_runtime/judges/claude.py +159 -0
- evalrx/agent_runtime/judges/codex.py +120 -0
- evalrx/agent_runtime/providers/__init__.py +21 -0
- evalrx/agent_runtime/providers/antigravity.py +31 -0
- evalrx/agent_runtime/providers/base.py +145 -0
- evalrx/agent_runtime/providers/claude_code.py +49 -0
- evalrx/agent_runtime/providers/codex.py +37 -0
- evalrx/agent_runtime/providers/gemini_cli.py +26 -0
- evalrx/agent_runtime/providers/kimi_cli.py +27 -0
- evalrx/agent_runtime/providers/opencode.py +27 -0
- evalrx/agent_runtime/providers/registry.py +58 -0
- evalrx/agent_runtime/sandbox.py +517 -0
- evalrx/agent_runtime/skill_audit.py +143 -0
- evalrx/agent_runtime/skills/__init__.py +19 -0
- evalrx/agent_runtime/skills/installer.py +68 -0
- evalrx/agent_runtime/skills/prompt_policy.py +86 -0
- evalrx/agent_runtime/skills/resolver.py +19 -0
- evalrx/analysis/__init__.py +132 -0
- evalrx/analysis/adjudicate.py +154 -0
- evalrx/analysis/analysis_module.py +361 -0
- evalrx/analysis/api.py +171 -0
- evalrx/analysis/case_studio.py +651 -0
- evalrx/analysis/cli.py +114 -0
- evalrx/analysis/dashboard.py +350 -0
- evalrx/analysis/eval_case_matrix.py +118 -0
- evalrx/analysis/eval_viz_theme.py +833 -0
- evalrx/analysis/explore_run.py +333 -0
- evalrx/analysis/explorer.py +1276 -0
- evalrx/analysis/failure_modes.py +607 -0
- evalrx/analysis/fused_pipeline.py +489 -0
- evalrx/analysis/holdout.py +300 -0
- evalrx/analysis/hypothesis_agent.py +230 -0
- evalrx/analysis/narration.py +177 -0
- evalrx/analysis/operationalize.py +442 -0
- evalrx/analysis/plain_language.py +42 -0
- evalrx/analysis/planner.py +283 -0
- evalrx/analysis/probe_search.py +203 -0
- evalrx/analysis/profile.py +268 -0
- evalrx/analysis/prompts/__init__.py +0 -0
- evalrx/analysis/prompts/explorer.py +417 -0
- evalrx/analysis/prompts/failure_modes.py +33 -0
- evalrx/analysis/prompts/holdout.py +27 -0
- evalrx/analysis/prompts/hypothesis_agent.py +78 -0
- evalrx/analysis/prompts/run_codebase.py +47 -0
- evalrx/analysis/prompts/stats_agent.py +72 -0
- evalrx/analysis/prompts/stats_tool_generator.py +43 -0
- evalrx/analysis/result_marker.py +47 -0
- evalrx/analysis/run_codebase.py +242 -0
- evalrx/analysis/run_view.py +205 -0
- evalrx/analysis/stage_views.py +93 -0
- evalrx/analysis/stats_agent.py +944 -0
- evalrx/analysis/stats_tool_agent.py +261 -0
- evalrx/analysis/stats_tool_generator.py +415 -0
- evalrx/analysis/stats_tools.py +1153 -0
- evalrx/analysis/trajectory_records.py +193 -0
- evalrx/analysis/workbench.py +431 -0
- evalrx/analyzers/__init__.py +42 -0
- evalrx/analyzers/agent/__init__.py +25 -0
- evalrx/analyzers/agent/counterfactual.py +84 -0
- evalrx/analyzers/agent/first_error_judge.py +96 -0
- evalrx/analyzers/agent/ignored_obs.py +81 -0
- evalrx/analyzers/agent/loop_detect.py +79 -0
- evalrx/analyzers/agent/reliability.py +165 -0
- evalrx/analyzers/agent/tool_shap.py +225 -0
- evalrx/analyzers/agent/trajectory_rubric.py +168 -0
- evalrx/analyzers/attention/__init__.py +19 -0
- evalrx/analyzers/attention/relative_attn.py +610 -0
- evalrx/analyzers/attention/rollout.py +73 -0
- evalrx/analyzers/attention/sink.py +56 -0
- evalrx/analyzers/attention/summary.py +190 -0
- evalrx/analyzers/attribution/__init__.py +6 -0
- evalrx/analyzers/attribution/generic_attn.py +31 -0
- evalrx/analyzers/attribution/gradcam.py +30 -0
- evalrx/analyzers/base.py +12 -0
- evalrx/analyzers/geometry/__init__.py +6 -0
- evalrx/analyzers/geometry/cka.py +70 -0
- evalrx/analyzers/geometry/linear_probe.py +157 -0
- evalrx/analyzers/hallucination/__init__.py +9 -0
- evalrx/analyzers/hallucination/chair.py +78 -0
- evalrx/analyzers/hallucination/opera.py +29 -0
- evalrx/analyzers/hallucination/pope.py +119 -0
- evalrx/analyzers/hallucination/selfcheck.py +155 -0
- evalrx/analyzers/hallucination/vcd.py +29 -0
- evalrx/analyzers/lens/__init__.py +7 -0
- evalrx/analyzers/lens/layer_contrast.py +133 -0
- evalrx/analyzers/lens/logit_lens.py +138 -0
- evalrx/analyzers/lens/tuned_lens.py +30 -0
- evalrx/analyzers/patching/__init__.py +5 -0
- evalrx/analyzers/patching/causal_trace.py +30 -0
- evalrx/analyzers/perturbation/__init__.py +23 -0
- evalrx/analyzers/perturbation/_shapley.py +54 -0
- evalrx/analyzers/perturbation/context_shap.py +174 -0
- evalrx/analyzers/perturbation/cot_faithfulness.py +239 -0
- evalrx/analyzers/perturbation/format_sensitivity.py +237 -0
- evalrx/analyzers/perturbation/mm_shap.py +146 -0
- evalrx/analyzers/perturbation/modality_ablation.py +196 -0
- evalrx/analyzers/perturbation/perturbation_battery.py +274 -0
- evalrx/analyzers/perturbation/prompt_contrast.py +265 -0
- evalrx/analyzers/perturbation/rise.py +94 -0
- evalrx/analyzers/perturbation/vl_shap.py +102 -0
- evalrx/analyzers/reasoning/__init__.py +33 -0
- evalrx/analyzers/reasoning/_text.py +328 -0
- evalrx/analyzers/reasoning/answer_extraction_audit.py +327 -0
- evalrx/analyzers/reasoning/arith_audit.py +226 -0
- evalrx/analyzers/reasoning/contamination.py +214 -0
- evalrx/analyzers/reasoning/knowledge_split.py +253 -0
- evalrx/analyzers/reasoning/self_repair.py +246 -0
- evalrx/analyzers/reasoning/step_rollout_value.py +216 -0
- evalrx/analyzers/reasoning/termination_audit.py +258 -0
- evalrx/analyzers/uncertainty/__init__.py +18 -0
- evalrx/analyzers/uncertainty/calibration.py +174 -0
- evalrx/analyzers/uncertainty/coverage_gap.py +199 -0
- evalrx/analyzers/uncertainty/entropy.py +90 -0
- evalrx/analyzers/uncertainty/logprob_entropy.py +69 -0
- evalrx/analyzers/uncertainty/self_consistency.py +204 -0
- evalrx/analyzers/uncertainty/verbalized_conf.py +64 -0
- evalrx/cli.py +411 -0
- evalrx/config.py +77 -0
- evalrx/contract/__init__.py +179 -0
- evalrx/contract/common.py +452 -0
- evalrx/contract/emit.py +948 -0
- evalrx/contract/export.py +237 -0
- evalrx/contract/m1.py +325 -0
- evalrx/contract/m2.py +317 -0
- evalrx/contract/m3.py +165 -0
- evalrx/contract/m4.py +130 -0
- evalrx/contract/m5.py +292 -0
- evalrx/contract/methodology.py +76 -0
- evalrx/contract/pre_m1.py +58 -0
- evalrx/contract/typescript.py +140 -0
- evalrx/core/__init__.py +85 -0
- evalrx/core/analyzer.py +174 -0
- evalrx/core/capability.py +54 -0
- evalrx/core/case.py +443 -0
- evalrx/core/experiment.py +106 -0
- evalrx/core/model.py +198 -0
- evalrx/core/pipeline.py +42 -0
- evalrx/core/registry.py +142 -0
- evalrx/core/result.py +64 -0
- evalrx/core/spec.py +173 -0
- evalrx/core/tokentype.py +165 -0
- evalrx/core/tool.py +92 -0
- evalrx/datasets/__init__.py +41 -0
- evalrx/datasets/base.py +68 -0
- evalrx/datasets/gui_os.py +52 -0
- evalrx/datasets/llm_qa.py +57 -0
- evalrx/datasets/pure_qa.py +12 -0
- evalrx/datasets/vlm_qa.py +695 -0
- evalrx/datasets/web_search_qa.py +52 -0
- evalrx/eval_agent/__init__.py +341 -0
- evalrx/eval_agent/_tools.py +81 -0
- evalrx/eval_agent/ab_runner.py +50 -0
- evalrx/eval_agent/agentic/__init__.py +43 -0
- evalrx/eval_agent/agentic/actions.py +216 -0
- evalrx/eval_agent/agentic/board.py +107 -0
- evalrx/eval_agent/agentic/loop.py +190 -0
- evalrx/eval_agent/agentic/tools.py +538 -0
- evalrx/eval_agent/checkpoint.py +57 -0
- evalrx/eval_agent/cli_agent.py +59 -0
- evalrx/eval_agent/cli_skills.py +5 -0
- evalrx/eval_agent/evolution.py +396 -0
- evalrx/eval_agent/git_manager.py +215 -0
- evalrx/eval_agent/hypothesis.py +172 -0
- evalrx/eval_agent/label_quarantine.py +209 -0
- evalrx/eval_agent/legacy.py +530 -0
- evalrx/eval_agent/log_schema.py +497 -0
- evalrx/eval_agent/loop.py +2159 -0
- evalrx/eval_agent/loop_reports.py +116 -0
- evalrx/eval_agent/model_instrumentation.py +282 -0
- evalrx/eval_agent/narration.py +193 -0
- evalrx/eval_agent/nl_runner.py +460 -0
- evalrx/eval_agent/orchestrator.py +61 -0
- evalrx/eval_agent/preregister.py +93 -0
- evalrx/eval_agent/prompts/__init__.py +1 -0
- evalrx/eval_agent/prompts/agentic.py +46 -0
- evalrx/eval_agent/prompts/case_discovery.py +25 -0
- evalrx/eval_agent/prompts/diagnosis.py +125 -0
- evalrx/eval_agent/prompts/experiment_writer.py +265 -0
- evalrx/eval_agent/prompts/explore_step.py +37 -0
- evalrx/eval_agent/prompts/fix_agent.py +257 -0
- evalrx/eval_agent/prompts/hypothesis_tester.py +15 -0
- evalrx/eval_agent/prompts/nl_runner.py +38 -0
- evalrx/eval_agent/prompts/probe_agent.py +25 -0
- evalrx/eval_agent/prompts/probe_candidate_generator.py +14 -0
- evalrx/eval_agent/prompts/probe_generator.py +35 -0
- evalrx/eval_agent/prompts/whitebox_probe_generator.py +38 -0
- evalrx/eval_agent/report.py +58 -0
- evalrx/eval_agent/run_context.py +354 -0
- evalrx/eval_agent/run_log.schema.json +1215 -0
- evalrx/eval_agent/run_logger_v2.py +1764 -0
- evalrx/eval_agent/run_metadata.py +208 -0
- evalrx/eval_agent/stages/__init__.py +56 -0
- evalrx/eval_agent/stages/case_discovery.py +293 -0
- evalrx/eval_agent/stages/diagnosis.py +1017 -0
- evalrx/eval_agent/stages/experiment_writer.py +1634 -0
- evalrx/eval_agent/stages/fix_agent.py +3916 -0
- evalrx/eval_agent/stages/fix_internals.py +499 -0
- evalrx/eval_agent/stages/fix_pipeline.py +725 -0
- evalrx/eval_agent/stages/fix_tiers.py +187 -0
- evalrx/eval_agent/stages/fix_tools.py +1034 -0
- evalrx/eval_agent/stages/hypothesis_tester.py +1014 -0
- evalrx/eval_agent/stages/probe.py +439 -0
- evalrx/eval_agent/stages/probe_agent.py +1079 -0
- evalrx/eval_agent/stages/probe_candidate_generator.py +128 -0
- evalrx/eval_agent/stages/probe_generator.py +326 -0
- evalrx/eval_agent/stages/probe_search_agent.py +106 -0
- evalrx/eval_agent/stages/protocol.py +112 -0
- evalrx/eval_agent/stages/repair_catalog.py +273 -0
- evalrx/eval_agent/stages/surgery.py +524 -0
- evalrx/eval_agent/stages/whitebox_probe_generator.py +351 -0
- evalrx/eval_agent/store.py +231 -0
- evalrx/logging_utils.py +112 -0
- evalrx/models/__init__.py +161 -0
- evalrx/models/_discover.py +101 -0
- evalrx/models/agent.py +380 -0
- evalrx/models/backends/__init__.py +58 -0
- evalrx/models/backends/api.py +169 -0
- evalrx/models/backends/base.py +57 -0
- evalrx/models/backends/gemini_compat.py +579 -0
- evalrx/models/backends/hf_local.py +2074 -0
- evalrx/models/backends/openai_compat.py +301 -0
- evalrx/models/backends/vllm_offline.py +116 -0
- evalrx/models/base.py +24 -0
- evalrx/models/blackbox/__init__.py +4 -0
- evalrx/models/blackbox/agent.py +31 -0
- evalrx/models/blackbox/base.py +29 -0
- evalrx/models/blackbox/gemini.py +279 -0
- evalrx/models/blackbox/llm_api.py +17 -0
- evalrx/models/blackbox/vlm_api.py +17 -0
- evalrx/models/compose.py +66 -0
- evalrx/models/inference.py +88 -0
- evalrx/models/paper_methods/__init__.py +8 -0
- evalrx/models/paper_methods/aad.py +53 -0
- evalrx/models/paper_methods/ifcd.py +204 -0
- evalrx/models/paper_methods/pai.py +164 -0
- evalrx/models/paper_methods/tcd.py +202 -0
- evalrx/models/paper_methods/vcd.py +45 -0
- evalrx/models/paper_methods/vicrop.py +137 -0
- evalrx/models/toolcodec.py +143 -0
- evalrx/models/tools/__init__.py +20 -0
- evalrx/models/tools/perception.py +300 -0
- evalrx/models/tools/visual.py +174 -0
- evalrx/models/whitebox/__init__.py +26 -0
- evalrx/models/whitebox/agent.py +31 -0
- evalrx/models/whitebox/base.py +24 -0
- evalrx/models/whitebox/qwen.py +61 -0
- evalrx/models/whitebox/qwen2_5_omni.py +29 -0
- evalrx/models/whitebox/qwen2_audio.py +25 -0
- evalrx/models/whitebox/qwen_omni.py +53 -0
- evalrx/models/whitebox/qwen_vl.py +62 -0
- evalrx/observability/__init__.py +21 -0
- evalrx/observability/envelope.py +122 -0
- evalrx/observability/outbox.py +111 -0
- evalrx/observability/tracer.py +882 -0
- evalrx/reporting/__init__.py +28 -0
- evalrx/reporting/case_study.py +947 -0
- evalrx/reporting/compiler.py +587 -0
- evalrx/reporting/dynamic.py +1882 -0
- evalrx/reporting/html_report.py +2225 -0
- evalrx/reporting/langfuse_exporter.py +38 -0
- evalrx/reporting/langfuse_source.py +155 -0
- evalrx/reporting/model.py +151 -0
- evalrx/reporting/run_events.py +184 -0
- evalrx/reporting/server.py +557 -0
- evalrx/reporting/stages.py +58 -0
- evalrx/reporting/static_export.py +142 -0
- evalrx/reporting/web_dist/index.html +146 -0
- evalrx/specs.py +727 -0
- evalrx/stats/__init__.py +47 -0
- evalrx/stats/api.py +192 -0
- evalrx/stats/bootstrap.py +86 -0
- evalrx/stats/ebh.py +27 -0
- evalrx/stats/evalue.py +98 -0
- evalrx/stats/friedman.py +138 -0
- evalrx/stats/mcnemar.py +40 -0
- evalrx/stats/multiplicity.py +159 -0
- evalrx/stats/subset_sampling.py +55 -0
- evalrx/term_links.py +43 -0
- evalrx/viz/__init__.py +7 -0
- evalrx/viz/labels.py +77 -0
- evalrx/viz/prompts.py +39 -0
- evalrx/viz/renderer.py +590 -0
- evalrx/viz/schema.py +36 -0
- evalrx/viz/style.py +134 -0
- evalrx-0.1.2.dist-info/METADATA +532 -0
- evalrx-0.1.2.dist-info/RECORD +339 -0
- evalrx-0.1.2.dist-info/WHEEL +5 -0
- evalrx-0.1.2.dist-info/entry_points.txt +3 -0
- evalrx-0.1.2.dist-info/licenses/LICENSE +121 -0
- evalrx-0.1.2.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,251 @@
|
|
|
1
|
+
# Tutorials — Nature Figure Making
|
|
2
|
+
|
|
3
|
+
End-to-end walkthroughs for the most common publication figure types.
|
|
4
|
+
All examples use helpers from [api.md](api.md) and patterns from [common-patterns.md](common-patterns.md).
|
|
5
|
+
For real production scripts and output previews from figures4papers, open [demos.md](demos.md).
|
|
6
|
+
|
|
7
|
+
---
|
|
8
|
+
|
|
9
|
+
## Tutorial 1: Grouped bar chart (multi-metric comparison)
|
|
10
|
+
|
|
11
|
+
**Goal**: Several methods compared across multiple metrics. Legend in a dedicated panel.
|
|
12
|
+
When methods belong to related families, use one coherent baseline family plus one coherent hero family.
|
|
13
|
+
|
|
14
|
+
```python
|
|
15
|
+
import os
|
|
16
|
+
import numpy as np
|
|
17
|
+
import matplotlib.pyplot as plt
|
|
18
|
+
from matplotlib import gridspec
|
|
19
|
+
|
|
20
|
+
# --- Style ---
|
|
21
|
+
plt.rcParams['font.family'] = 'sans-serif'
|
|
22
|
+
plt.rcParams['font.sans-serif'] = ['Arial']
|
|
23
|
+
plt.rcParams['svg.fonttype'] = 'none'
|
|
24
|
+
plt.rcParams['font.size'] = 24
|
|
25
|
+
plt.rcParams['axes.spines.right'] = False
|
|
26
|
+
plt.rcParams['axes.spines.top'] = False
|
|
27
|
+
plt.rcParams['axes.linewidth'] = 3
|
|
28
|
+
|
|
29
|
+
# --- Data ---
|
|
30
|
+
methods = ['ResNet1d18', 'ResNet1d34', 'ECGFounder', 'CSFM-Tiny', 'CSFM-Base', 'CSFM-Large']
|
|
31
|
+
colors = ['#484878', '#7884B4', '#B4C0E4', '#E4E4F0', '#E4CCD8', '#F0C0CC']
|
|
32
|
+
metrics = ['Metric 1', 'Metric 2', 'Metric 3']
|
|
33
|
+
mean = {
|
|
34
|
+
'Metric 1': np.array([0.81, 0.83, 0.86, 0.89, 0.91, 0.92]),
|
|
35
|
+
'Metric 2': np.array([0.63, 0.67, 0.71, 0.74, 0.77, 0.79]),
|
|
36
|
+
'Metric 3': np.array([0.41, 0.45, 0.49, 0.53, 0.56, 0.58]),
|
|
37
|
+
}
|
|
38
|
+
std = {k: v * 0.03 for k, v in mean.items()} # placeholder
|
|
39
|
+
|
|
40
|
+
# --- Figure ---
|
|
41
|
+
fig = plt.figure(figsize=(28, 6))
|
|
42
|
+
gs = gridspec.GridSpec(1, len(metrics) + 1) # +1 for legend panel
|
|
43
|
+
|
|
44
|
+
handles, labels = None, None
|
|
45
|
+
for col, metric in enumerate(metrics):
|
|
46
|
+
ax = fig.add_subplot(gs[col])
|
|
47
|
+
bars = ax.bar(
|
|
48
|
+
range(len(methods)),
|
|
49
|
+
mean[metric],
|
|
50
|
+
yerr=std[metric],
|
|
51
|
+
capsize=5,
|
|
52
|
+
color=colors,
|
|
53
|
+
label=methods,
|
|
54
|
+
error_kw={'elinewidth': 2, 'capthick': 2},
|
|
55
|
+
)
|
|
56
|
+
if col == 0:
|
|
57
|
+
handles, labels = ax.get_legend_handles_labels()
|
|
58
|
+
ax.set_xticks([])
|
|
59
|
+
y_vals = mean[metric]
|
|
60
|
+
margin = (y_vals.max() - y_vals.min()) * 0.15
|
|
61
|
+
ax.set_ylim([y_vals.min() - margin, y_vals.max() + margin])
|
|
62
|
+
ax.set_ylabel(metric, fontsize=32)
|
|
63
|
+
|
|
64
|
+
# Legend-only panel
|
|
65
|
+
ax_leg = fig.add_subplot(gs[-1])
|
|
66
|
+
ax_leg.legend(handles, labels, fontsize=28, loc='center', frameon=False)
|
|
67
|
+
ax_leg.set_axis_off()
|
|
68
|
+
|
|
69
|
+
fig.tight_layout(pad=2)
|
|
70
|
+
os.makedirs('./figures', exist_ok=True)
|
|
71
|
+
fig.savefig('./figures/comparison.png', dpi=300)
|
|
72
|
+
fig.savefig('./figures/comparison.pdf', dpi=300)
|
|
73
|
+
plt.close(fig)
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
---
|
|
77
|
+
|
|
78
|
+
## Tutorial 2: Ablation bar chart (alpha-graduated, horizontal)
|
|
79
|
+
|
|
80
|
+
**Goal**: Same method with components progressively added; alpha encodes completeness.
|
|
81
|
+
|
|
82
|
+
```python
|
|
83
|
+
import os
|
|
84
|
+
import numpy as np
|
|
85
|
+
import matplotlib.pyplot as plt
|
|
86
|
+
|
|
87
|
+
plt.rcParams['font.family'] = 'sans-serif'
|
|
88
|
+
plt.rcParams['font.sans-serif'] = ['Arial']
|
|
89
|
+
plt.rcParams['svg.fonttype'] = 'none'
|
|
90
|
+
plt.rcParams['font.size'] = 24
|
|
91
|
+
plt.rcParams['axes.spines.right'] = False
|
|
92
|
+
plt.rcParams['axes.spines.top'] = False
|
|
93
|
+
plt.rcParams['axes.linewidth'] = 3
|
|
94
|
+
|
|
95
|
+
configs = ['None', '+ Module A', '+ Module B', '+ Module C', 'Full']
|
|
96
|
+
values = np.array([0.72, 0.78, 0.81, 0.84, 0.88])
|
|
97
|
+
stds = np.array([0.02, 0.02, 0.01, 0.01, 0.01])
|
|
98
|
+
|
|
99
|
+
n = len(configs)
|
|
100
|
+
blue_rgb = (0.215686, 0.458824, 0.729412) # #3775BA
|
|
101
|
+
alphas = np.linspace(0.2, 1.0, n)
|
|
102
|
+
colors = [(blue_rgb[0], blue_rgb[1], blue_rgb[2], a) for a in alphas]
|
|
103
|
+
|
|
104
|
+
fig, ax = plt.subplots(figsize=(12, 6))
|
|
105
|
+
ax.barh(range(n), values, xerr=stds,
|
|
106
|
+
color=colors, ecolor='k', capsize=5)
|
|
107
|
+
ax.set_yticks(range(n))
|
|
108
|
+
ax.set_yticklabels(configs)
|
|
109
|
+
ax.set_xlim([values.min() - 0.05, values.max() + 0.03])
|
|
110
|
+
ax.set_xlabel('Score', fontsize=32)
|
|
111
|
+
|
|
112
|
+
fig.tight_layout(pad=2)
|
|
113
|
+
os.makedirs('./figures', exist_ok=True)
|
|
114
|
+
fig.savefig('./figures/ablation.png', dpi=300)
|
|
115
|
+
plt.close(fig)
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
---
|
|
119
|
+
|
|
120
|
+
## Tutorial 3: Multi-panel trend with shared legend
|
|
121
|
+
|
|
122
|
+
**Goal**: Two trend panels (e.g., train/val curves) and a legend-only third panel.
|
|
123
|
+
|
|
124
|
+
```python
|
|
125
|
+
import os
|
|
126
|
+
import numpy as np
|
|
127
|
+
import matplotlib.pyplot as plt
|
|
128
|
+
|
|
129
|
+
plt.rcParams['font.family'] = 'sans-serif'
|
|
130
|
+
plt.rcParams['font.sans-serif'] = ['Arial']
|
|
131
|
+
plt.rcParams['svg.fonttype'] = 'none'
|
|
132
|
+
plt.rcParams['font.size'] = 15
|
|
133
|
+
plt.rcParams['axes.spines.right'] = False
|
|
134
|
+
plt.rcParams['axes.spines.top'] = False
|
|
135
|
+
plt.rcParams['axes.linewidth'] = 2
|
|
136
|
+
|
|
137
|
+
methods = ['Baseline', 'CSFM-Tiny', 'CSFM-Base', 'CSFM-Large']
|
|
138
|
+
colors = ['#7884B4', '#E4E4F0', '#E4CCD8', '#F0C0CC']
|
|
139
|
+
x = np.arange(0, 100, 5)
|
|
140
|
+
|
|
141
|
+
fig, axes = plt.subplots(1, 3, figsize=(18, 5))
|
|
142
|
+
|
|
143
|
+
for panel_idx, (ax, panel_name) in enumerate(zip(axes[:2], ['Training', 'Validation'])):
|
|
144
|
+
for method, color in zip(methods, colors):
|
|
145
|
+
y = 0.48 + 0.42 * (1 - np.exp(-x / 30)) + np.random.randn(len(x)) * 0.01
|
|
146
|
+
if method == 'Baseline':
|
|
147
|
+
y -= 0.03
|
|
148
|
+
elif method == 'CSFM-Tiny':
|
|
149
|
+
y += 0.00
|
|
150
|
+
elif method == 'CSFM-Base':
|
|
151
|
+
y += 0.02
|
|
152
|
+
elif method == 'CSFM-Large':
|
|
153
|
+
y += 0.03
|
|
154
|
+
ax.plot(x, y, color=color, lw=2.5, marker='o', markersize=6, label=method)
|
|
155
|
+
ax.set_title(panel_name, fontsize=18)
|
|
156
|
+
ax.set_xlabel('Epoch', fontsize=16)
|
|
157
|
+
ax.set_ylabel('Loss', fontsize=16)
|
|
158
|
+
if panel_idx == 0:
|
|
159
|
+
handles, labels = ax.get_legend_handles_labels()
|
|
160
|
+
|
|
161
|
+
# Legend-only panel
|
|
162
|
+
axes[2].legend(handles, labels, fontsize=14, loc='center', frameon=False)
|
|
163
|
+
axes[2].set_axis_off()
|
|
164
|
+
|
|
165
|
+
fig.tight_layout(pad=2)
|
|
166
|
+
os.makedirs('./figures', exist_ok=True)
|
|
167
|
+
fig.savefig('./figures/trends.png', dpi=300)
|
|
168
|
+
fig.savefig('./figures/trends.pdf', dpi=300)
|
|
169
|
+
plt.close(fig)
|
|
170
|
+
```
|
|
171
|
+
|
|
172
|
+
---
|
|
173
|
+
|
|
174
|
+
## Tutorial 4: Heatmap with dual colormaps (positive/negative columns)
|
|
175
|
+
|
|
176
|
+
**Goal**: Score matrix where positive = Reds, negative = Blues_r. Cell text auto-contrasted.
|
|
177
|
+
|
|
178
|
+
```python
|
|
179
|
+
import os
|
|
180
|
+
import numpy as np
|
|
181
|
+
import matplotlib as mpl
|
|
182
|
+
import matplotlib.pyplot as plt
|
|
183
|
+
|
|
184
|
+
plt.rcParams['font.family'] = 'sans-serif'
|
|
185
|
+
plt.rcParams['font.sans-serif'] = ['Arial']
|
|
186
|
+
plt.rcParams['svg.fonttype'] = 'none'
|
|
187
|
+
plt.rcParams['font.size'] = 16
|
|
188
|
+
plt.rcParams['axes.spines.right'] = False
|
|
189
|
+
plt.rcParams['axes.spines.top'] = False
|
|
190
|
+
plt.rcParams['axes.linewidth'] = 2
|
|
191
|
+
|
|
192
|
+
# matrix: rows = methods, cols = metrics (alternating positive/negative directions)
|
|
193
|
+
methods = ['Method A', 'Method B', 'Method C', 'Method D']
|
|
194
|
+
metrics = ['Score (+)', 'Error (-)', 'F1 (+)', 'Loss (-)']
|
|
195
|
+
matrix = np.array([
|
|
196
|
+
[0.88, 0.12, 0.85, 0.20],
|
|
197
|
+
[0.81, 0.18, 0.78, 0.28],
|
|
198
|
+
[0.75, 0.25, 0.72, 0.35],
|
|
199
|
+
[0.70, 0.30, 0.68, 0.40],
|
|
200
|
+
])
|
|
201
|
+
|
|
202
|
+
fig, ax = plt.subplots(figsize=(10, 6))
|
|
203
|
+
n_rows, n_cols = matrix.shape
|
|
204
|
+
vmin, vmax = matrix.min(0), matrix.max(0)
|
|
205
|
+
|
|
206
|
+
for j in range(n_cols):
|
|
207
|
+
is_positive = (j % 2 == 0)
|
|
208
|
+
cmap = plt.cm.Reds if is_positive else plt.cm.Blues_r
|
|
209
|
+
cmap = cmap.copy()
|
|
210
|
+
norm = mpl.colors.Normalize(
|
|
211
|
+
vmin=0 if is_positive else vmax[j],
|
|
212
|
+
vmax=vmax[j] if is_positive else 0
|
|
213
|
+
)
|
|
214
|
+
ax.imshow(matrix[:, j:j+1], cmap=cmap, norm=norm,
|
|
215
|
+
aspect='auto', extent=[j-0.5, j+0.5, 0, n_rows], origin='lower')
|
|
216
|
+
|
|
217
|
+
for (i, j), val in np.ndenumerate(matrix):
|
|
218
|
+
is_positive = (j % 2 == 0)
|
|
219
|
+
cmap = plt.cm.Reds if is_positive else plt.cm.Blues_r
|
|
220
|
+
norm = mpl.colors.Normalize(vmin=0 if is_positive else vmax[j],
|
|
221
|
+
vmax=vmax[j] if is_positive else 0)
|
|
222
|
+
r, g, b, _ = cmap(norm(val))
|
|
223
|
+
lum = 0.299*r + 0.587*g + 0.114*b
|
|
224
|
+
color = 'white' if lum < 0.5 else 'black'
|
|
225
|
+
ax.text(j, i + 0.5, f'{val:.2f}', ha='center', va='center',
|
|
226
|
+
fontsize=13, color=color)
|
|
227
|
+
|
|
228
|
+
ax.set_xlim(-0.5, n_cols - 0.5)
|
|
229
|
+
ax.set_xticks(np.arange(n_cols))
|
|
230
|
+
ax.set_xticklabels(metrics, rotation=30, ha='right', fontsize=14)
|
|
231
|
+
ax.tick_params(axis='x', bottom=False, top=False, length=0)
|
|
232
|
+
ax.set_yticks(np.arange(n_rows) + 0.5)
|
|
233
|
+
ax.set_yticklabels(methods, fontsize=14)
|
|
234
|
+
ax.set_frame_on(False)
|
|
235
|
+
ax.invert_yaxis()
|
|
236
|
+
|
|
237
|
+
fig.tight_layout(pad=2)
|
|
238
|
+
os.makedirs('./figures', exist_ok=True)
|
|
239
|
+
fig.savefig('./figures/heatmap.png', dpi=300)
|
|
240
|
+
plt.close(fig)
|
|
241
|
+
```
|
|
242
|
+
|
|
243
|
+
---
|
|
244
|
+
|
|
245
|
+
## Related files
|
|
246
|
+
|
|
247
|
+
- [SKILL.md](../SKILL.md) — When to use this skill
|
|
248
|
+
- [api.md](api.md) — Reusable helper implementations
|
|
249
|
+
- [common-patterns.md](common-patterns.md) — Layout and encoding patterns used above
|
|
250
|
+
- [design-theory.md](design-theory.md) — Why these choices exist
|
|
251
|
+
- [chart-types.md](chart-types.md) — Radar, 3D sphere, scatter, fill_between
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
# Figure contract before plotting
|
|
2
|
+
|
|
3
|
+
A publication-quality scientific figure is a visual argument, not an isolated pretty plot. Every figure starts from a claim, an evidence hierarchy, and a review-risk check before code or aesthetics. Before generating or editing code, establish the contract below.
|
|
4
|
+
|
|
5
|
+
## Backend selection is a blocking gate
|
|
6
|
+
|
|
7
|
+
If the user has not explicitly chosen Python or R in the current request or provided a clearly language-specific input file/workflow, ask one concise question: **Python or R?** Then stop and wait for the user's answer. Do not generate mock data, write scripts, create figures, or choose Python/R by default. This overrides general autonomy/default-execution behavior for figure tasks.
|
|
8
|
+
|
|
9
|
+
Only recommend a backend when the user explicitly asks you to choose or recommend one. In that case, use `references/backend-selection.md`, state the reason, and then proceed with the recommended backend.
|
|
10
|
+
|
|
11
|
+
## The selected backend is exclusive
|
|
12
|
+
|
|
13
|
+
Once Python or R is selected, every plotting script, preview image, SVG/PDF/TIFF/PNG export, QA render, and visual workaround must be produced by that same backend. Do not use Python to draw a preview for an R figure, and do not use R to draw a preview for a Python figure, even if the selected runtime or packages are missing locally. The non-selected language may only be used for non-visual file inspection or data conversion when it does not open a graphics device, import plotting libraries, create image/vector files, or change the final visual appearance.
|
|
14
|
+
|
|
15
|
+
## Missing runtime/package rule
|
|
16
|
+
|
|
17
|
+
After the backend is selected, check the selected runtime early (`Rscript`/R for R; Python and required plotting packages for Python). If the selected runtime or required packages are unavailable, stop before rendering and report the exact blocker. You may provide a selected-backend script and installation commands, or ask permission to install dependencies, but you must not fall back to the other language to make a substitute figure.
|
|
18
|
+
|
|
19
|
+
## The five-point contract
|
|
20
|
+
|
|
21
|
+
1. **Core conclusion**: write the one-sentence claim the figure must defend.
|
|
22
|
+
2. **Evidence chain**: map each planned panel to the claim, and drop panels that do not carry a unique piece of evidence.
|
|
23
|
+
3. **Archetype**: classify the figure as `quantitative grid`, `schematic-led composite`, `image plate + quant`, or `asymmetric mixed-modality figure`.
|
|
24
|
+
4. **Backend**: use the selected Python or R track exclusively for all figure drawing, previewing, exporting, and visual QA. Do not cross-render with the other language.
|
|
25
|
+
5. **Journal/export contract**: set final dimensions, editable text, source data, statistics, image-integrity notes, and export formats before styling.
|
|
26
|
+
|
|
27
|
+
The highest-priority rule is: **the chart serves the scientific logic**. Aesthetic polish, template matching, and complex layout are subordinate to making the core conclusion clear, defensible, and reviewable.
|
|
28
|
+
|
|
29
|
+
For the full method to convert a request into core conclusion, evidence hierarchy, panel map, and review-risk checks, open `references/figure-contract.md`.
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
# Default operating stance
|
|
2
|
+
|
|
3
|
+
The older Python/matplotlib rules in this skill remain valid. The skill also supports R, especially `ggplot2 + patchwork + ComplexHeatmap + ggrepel + svglite/cairo_pdf + ragg`.
|
|
4
|
+
|
|
5
|
+
## Color policy
|
|
6
|
+
|
|
7
|
+
Prefer **unified method families across all panels** over maximal hue separation. For dense Nature Machine Intelligence-style figure pages, use the low-saturation `NMI pastel` family described in `references/api.md` and reserve green/red mainly for gains, drops, and other directional cues.
|
|
8
|
+
|
|
9
|
+
## Stance
|
|
10
|
+
|
|
11
|
+
- Start by classifying the requested figure into one of four archetypes: `quantitative grid`, `schematic-led composite`, `image plate + quant`, or `asymmetric mixed-modality figure`.
|
|
12
|
+
- Prefer one **hero panel** plus subordinate evidence panels over filling the canvas with equal-sized subplots.
|
|
13
|
+
- If the user asks for a single chart, still identify its role in the manuscript claim: discovery, mechanism, validation, comparison, robustness, or clinical/biological relevance.
|
|
14
|
+
- Keep the background white for plots and diagrams; switch to black only for microscopy / volume-rendering image plates.
|
|
15
|
+
- Prefer direct labels over legends when categories are spatially fixed or the legend would force unnecessary eye travel.
|
|
16
|
+
- Keep one restrained palette per figure: usually one neutral family, one signal family, and one accent family.
|
|
17
|
+
- Treat statistics, `n`, error-bar definitions, source-data traceability, and image-integrity notes as part of the figure, not as optional caption cleanup.
|
|
18
|
+
- When the user asks for broad `Nature` style rather than ML/NMI-specific style, read `references/nature-2026-observations.md` before choosing layout.
|
|
19
|
+
- When the user references `figures4papers` or the older `scientific-figure-making` skill, treat this skill as the successor and open `references/demos.md` for bundled Python demo scripts.
|
|
20
|
+
|
|
21
|
+
## User-facing privacy rule
|
|
22
|
+
|
|
23
|
+
Do not disclose private local paths, private filenames, chat-attachment names, internal reference filenames, template identifiers, or the provenance of private working materials in user-facing replies, generated code comments, figure legends, reports, or manuscript text. Use generic descriptions such as "the provided R template collection", "a private working draft", or "the internal figure contract". If the user provides a private plotting template collection, use it only as an internal adaptation source and do not reveal its path, filenames, or provenance. Only reveal an exact path or source file when the user explicitly asks for that audit trail.
|
|
24
|
+
|
|
25
|
+
## When to load this skill
|
|
26
|
+
|
|
27
|
+
- Python or R figures for **papers, slides, or reports** targeting Nature, Science, Cell, NeurIPS, ICLR, or similar venues.
|
|
28
|
+
- Requests involving **grouped bars, trend lines, heatmaps, radar plots, multi-panel grids**, or **PDF/SVG/high-DPI** output.
|
|
29
|
+
- Any mention of "Nature style", "publication figure", "paper figure", "SCI figure", "figures4papers", "scientific-figure-making", "R plotting template", or "high-quality scientific plot".
|
|
30
|
+
- Requests to improve a figure's logic, aesthetics, panel layout, figure legend, export quality, or journal-readiness.
|
|
31
|
+
|
|
32
|
+
## When NOT to load
|
|
33
|
+
|
|
34
|
+
- Plotly, Altair, Bokeh, or other interactive/web-first plotting.
|
|
35
|
+
- EDA-only plots without a publication target.
|
|
36
|
+
- Primary workflow is 3D, GIS, or non-scientific illustration tooling.
|
|
37
|
+
- Illustrator / Figma–first layout.
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
# Backend: Python (matplotlib / seaborn)
|
|
2
|
+
|
|
3
|
+
**Python-only execution rule.** When the user has selected Python, do all figure drawing, previewing, exporting, and visual QA in Python. Do not call R/ggplot2, ComplexHeatmap, patchwork, or any R graphics device to create a temporary preview, fallback export, or layout approximation. If Python or required Python plotting packages are missing, stop before rendering and report the missing dependency. You may still write the Python script, provide `pip`/environment install commands, or ask permission to install dependencies, but do not cross-render the figure in R.
|
|
4
|
+
|
|
5
|
+
## Python quick-start
|
|
6
|
+
|
|
7
|
+
```python
|
|
8
|
+
import matplotlib as mpl
|
|
9
|
+
import matplotlib.pyplot as plt
|
|
10
|
+
|
|
11
|
+
mpl.rcParams.update({
|
|
12
|
+
"font.family": "sans-serif",
|
|
13
|
+
"font.sans-serif": ["Arial", "Helvetica", "DejaVu Sans", "sans-serif"],
|
|
14
|
+
"svg.fonttype": "none", # editable text in SVG
|
|
15
|
+
"pdf.fonttype": 42, # editable TrueType text in PDF
|
|
16
|
+
"font.size": 7, # use 15-24 only for large slide-sized panels
|
|
17
|
+
"axes.spines.right": False,
|
|
18
|
+
"axes.spines.top": False,
|
|
19
|
+
"axes.linewidth": 0.8,
|
|
20
|
+
"legend.frameon": False,
|
|
21
|
+
})
|
|
22
|
+
|
|
23
|
+
def save_pub_py(fig, filename, dpi=600):
|
|
24
|
+
fig.savefig(f"{filename}.svg", bbox_inches="tight")
|
|
25
|
+
fig.savefig(f"{filename}.pdf", bbox_inches="tight")
|
|
26
|
+
fig.savefig(f"{filename}.tiff", dpi=dpi, bbox_inches="tight")
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
Use `text.usetex = True` only when LaTeX is installed and math-rich labels are required.
|
|
30
|
+
|
|
31
|
+
## Going deeper
|
|
32
|
+
|
|
33
|
+
- `references/api.md` — Python PALETTE, helper function signatures, validation rules.
|
|
34
|
+
- `references/common-patterns.md` — hero panels, legend-only axes, dark image plates, asymmetric layouts.
|
|
35
|
+
- `references/chart-types.md` — radar, 3D sphere, fill_between, scatter patterns.
|
|
36
|
+
- `references/tutorials.md` — end-to-end walkthroughs for bars, trends, heatmaps.
|
|
37
|
+
- `references/demos.md` — bundled figures4papers Python scripts and output previews.
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
# Backend: R (ggplot2 / patchwork / ComplexHeatmap)
|
|
2
|
+
|
|
3
|
+
**R-only execution rule.** When the user has selected R, do all figure drawing, previewing, exporting, and visual QA in R. Do not call matplotlib/seaborn or any Python graphics device to create a temporary preview, fallback export, or layout approximation. If `Rscript`/R or required R packages are missing, stop before rendering and report the exact blocker. You may still write the R script, provide install commands (for example `install.packages(...)`), or ask permission to install dependencies, but do not cross-render the figure in Python.
|
|
4
|
+
|
|
5
|
+
## R quick-start
|
|
6
|
+
|
|
7
|
+
```r
|
|
8
|
+
library(ggplot2)
|
|
9
|
+
library(patchwork)
|
|
10
|
+
|
|
11
|
+
theme_set(
|
|
12
|
+
theme_classic(base_size = 6.5, base_family = "Arial") +
|
|
13
|
+
theme(
|
|
14
|
+
axis.line = element_line(linewidth = 0.35, colour = "black"),
|
|
15
|
+
axis.ticks = element_line(linewidth = 0.35, colour = "black"),
|
|
16
|
+
legend.title = element_text(size = 6.2),
|
|
17
|
+
legend.text = element_text(size = 5.8),
|
|
18
|
+
strip.text = element_text(size = 6.2, face = "bold"),
|
|
19
|
+
plot.title = element_text(size = 7, face = "bold"),
|
|
20
|
+
panel.grid = element_blank()
|
|
21
|
+
)
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
save_pub_r <- function(plot, filename, width_mm = 183, height_mm = 120, dpi = 600) {
|
|
25
|
+
w <- width_mm / 25.4
|
|
26
|
+
h <- height_mm / 25.4
|
|
27
|
+
svglite::svglite(paste0(filename, ".svg"), width = w, height = h)
|
|
28
|
+
print(plot)
|
|
29
|
+
dev.off()
|
|
30
|
+
grDevices::cairo_pdf(paste0(filename, ".pdf"), width = w, height = h, family = "Arial")
|
|
31
|
+
print(plot)
|
|
32
|
+
dev.off()
|
|
33
|
+
ragg::agg_tiff(paste0(filename, ".tiff"), width = w, height = h, units = "in", res = dpi)
|
|
34
|
+
print(plot)
|
|
35
|
+
dev.off()
|
|
36
|
+
}
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
## Going deeper
|
|
40
|
+
|
|
41
|
+
- `references/r-workflow.md` — the R plotting workflow when the user provides R scripts, templates, or data.
|
|
42
|
+
- `references/r-template-index.md` — adapt a user-provided or private R template collection without exposing source paths.
|
|
43
|
+
- `references/design-theory.md` — typography, color theory, layout rationale, export policy (backend-agnostic).
|
|
44
|
+
- `references/nature-2026-observations.md` — real Nature page archetypes to match before choosing layout.
|
|
@@ -0,0 +1,213 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: outcome-driver-analysis
|
|
3
|
+
description: >
|
|
4
|
+
Run a full, disciplined statistical analysis to find what differentiates a binary
|
|
5
|
+
outcome (success vs. fail, pass vs. fail, correct vs. incorrect) using a set of
|
|
6
|
+
explanatory variables the user specifies. Covers exploring the explanatory variables
|
|
7
|
+
themselves (distributions, outliers, missingness, correlations), exploring each
|
|
8
|
+
variable against the outcome (contingency tables for categorical variables,
|
|
9
|
+
distribution comparisons for continuous variables, conditioning on other variables),
|
|
10
|
+
marginal screening, fitting a justified regression model (logistic vs. linear vs.
|
|
11
|
+
mixed-effects, with explicit reasoning), goodness-of-fit diagnostics, result
|
|
12
|
+
visualization, and a plain-language written report. Use this whenever the user has
|
|
13
|
+
examples labeled by a binary outcome plus candidate explanatory variables and wants
|
|
14
|
+
to know what drives the difference -- e.g. analyzing model error cases ("why do
|
|
15
|
+
these examples fail"), pass/fail experiment results, or any dataset framed as
|
|
16
|
+
success/failure with covariates -- even if they don't use the word "statistics."
|
|
17
|
+
Do NOT use this for pure simulation studies with no real data, generating
|
|
18
|
+
paper-ready LaTeX tables for an already-written paper, or benchmarking many methods
|
|
19
|
+
against each other across datasets -- those are out of scope.
|
|
20
|
+
version: 0.2.0
|
|
21
|
+
license: MIT
|
|
22
|
+
compatibility: Claude Code project-scoped skill. Assumes R and/or Python available on PATH; pick whichever is appropriate per task rather than requiring both.
|
|
23
|
+
metadata:
|
|
24
|
+
tags: statistics research-workflow eda logistic-regression mixed-effects reproducibility R Python
|
|
25
|
+
agentskills_spec: "1.0"
|
|
26
|
+
---
|
|
27
|
+
|
|
28
|
+
# Outcome driver analysis
|
|
29
|
+
|
|
30
|
+
You are running a statistical analysis to explain a binary outcome (success/fail,
|
|
31
|
+
pass/fail, correct/incorrect) using explanatory variables the user provides. Work
|
|
32
|
+
through the pipeline below in order -- each step's findings feed the next one. Do not
|
|
33
|
+
skip straight to modeling: the exploratory steps surface issues (outliers, missing
|
|
34
|
+
data, collinearity, confounding) that change how the model should be built and read.
|
|
35
|
+
|
|
36
|
+
## Scope
|
|
37
|
+
|
|
38
|
+
This skill is for analyzing a real dataset with a binary outcome and candidate
|
|
39
|
+
explanatory variables -- most often model error analysis (why did these examples
|
|
40
|
+
fail?), but equally applicable to any pass/fail or success/failure outcome with
|
|
41
|
+
covariates. It is not for simulation studies with no real data, generating
|
|
42
|
+
publication-ready LaTeX tables for a paper that's already written, or benchmarking
|
|
43
|
+
many methods against each other across datasets.
|
|
44
|
+
|
|
45
|
+
## Chart types and palettes: defer to a chart-style skill when present
|
|
46
|
+
|
|
47
|
+
This skill decides WHAT to visualize at each step (the statistical intent);
|
|
48
|
+
it does not own HOW. When a dedicated chart-style skill (e.g.
|
|
49
|
+
`eval-chart-style`) is installed alongside this one, that skill's chart-type
|
|
50
|
+
policy and palette GOVERN every figure called for below — read the concrete
|
|
51
|
+
figure prescriptions in steps 2, 3 and 7 as the standalone fallback for when
|
|
52
|
+
no chart-style skill is present, never as an override of one.
|
|
53
|
+
|
|
54
|
+
## Choosing R or Python
|
|
55
|
+
|
|
56
|
+
Pick per task, in this order:
|
|
57
|
+
1. If the user is continuing an existing project, match whatever language that
|
|
58
|
+
project's code already uses.
|
|
59
|
+
2. Otherwise, pick based on the modeling need identified in step 5: mixed-effects /
|
|
60
|
+
hierarchical models are frequently more ergonomic in R (`lme4`, `glmmTMB`), while
|
|
61
|
+
plain logistic regression, screening, and diagnostics are equally well supported in
|
|
62
|
+
both R and Python (`statsmodels`, `scikit-learn`).
|
|
63
|
+
3. If genuinely ambiguous, ask the user.
|
|
64
|
+
|
|
65
|
+
Bundled helper scripts exist in both languages under `scripts/` -- use the one that
|
|
66
|
+
matches your choice; they are templates to adapt to the actual variable names and
|
|
67
|
+
data, not black boxes to run unmodified.
|
|
68
|
+
|
|
69
|
+
## 1. Intake -- clarify variables and structure
|
|
70
|
+
|
|
71
|
+
Before writing any code, confirm (from what the user already said, or by asking):
|
|
72
|
+
- Which column is the binary outcome, and which columns are the explanatory variables
|
|
73
|
+
to investigate.
|
|
74
|
+
- The research question or problem behind the analysis -- what decision or
|
|
75
|
+
understanding this is meant to support. Record it; the final report must be framed
|
|
76
|
+
around it, not written as a generic statistical summary.
|
|
77
|
+
- Each explanatory variable's type: categorical, continuous, or count.
|
|
78
|
+
- Whether observations are clustered or repeated -- e.g. multiple examples from the
|
|
79
|
+
same underlying model, prompt template, dataset, or subject. This is needed later to
|
|
80
|
+
decide whether a mixed-effects model is warranted.
|
|
81
|
+
|
|
82
|
+
## 2. Explanatory-variable EDA
|
|
83
|
+
|
|
84
|
+
Before relating anything to the outcome, characterize the explanatory variables on
|
|
85
|
+
their own terms:
|
|
86
|
+
- **Distribution**: histogram (continuous) or bar chart (categorical) per variable.
|
|
87
|
+
- **Outliers**: flag with an IQR or z-score rule for continuous variables; flag rare
|
|
88
|
+
categories for categorical variables.
|
|
89
|
+
- **Missingness**: count and pattern per variable. Spot-check whether missingness
|
|
90
|
+
itself is associated with the outcome (missing values are often not random).
|
|
91
|
+
- **Inter-variable structure**: correlation matrix for continuous-continuous pairs,
|
|
92
|
+
Cramer's V for categorical-categorical pairs, correlation ratio (eta-squared) or
|
|
93
|
+
ANOVA for mixed pairs. This surfaces redundant or collinear explanatory variables
|
|
94
|
+
early, before they cause multicollinearity problems at the modeling stage.
|
|
95
|
+
|
|
96
|
+
Use `scripts/explanatory_var_eda.R` or `.py` as a starting template.
|
|
97
|
+
|
|
98
|
+
## 3. Per-variable exploration vs. outcome
|
|
99
|
+
|
|
100
|
+
For each explanatory variable, relate it to the outcome:
|
|
101
|
+
- **Categorical variable** -> contingency table (variable x outcome) with row/column
|
|
102
|
+
proportions; chi-square test of independence, or Fisher's exact test when any
|
|
103
|
+
expected cell count is small (below ~5).
|
|
104
|
+
- **Continuous variable** -> a per-group distribution-comparison figure
|
|
105
|
+
(standalone fallback: side-by-side boxplot; an installed chart-style skill's
|
|
106
|
+
distribution-first policy — e.g. violin + jittered points — takes
|
|
107
|
+
precedence), plus a group-comparison test: Welch's t-test if roughly normal,
|
|
108
|
+
Mann-Whitney U as the robust default when normality is doubtful.
|
|
109
|
+
- Always report an effect size next to the test, not just a p-value: Cramer's V or
|
|
110
|
+
odds ratio for categorical variables, Cohen's d or rank-biserial correlation for
|
|
111
|
+
continuous variables.
|
|
112
|
+
- **Conditioning**: repeat the above stratified by, or faceted on, a plausible
|
|
113
|
+
confounding or interacting variable. Flag relationships that appear, vanish, or
|
|
114
|
+
reverse within strata (a Simpson's-paradox pattern) -- these are candidates for an
|
|
115
|
+
interaction term or control variable in the model stage, not things to quietly drop.
|
|
116
|
+
|
|
117
|
+
Use `scripts/univariate_eda.R` or `.py` as a starting template.
|
|
118
|
+
|
|
119
|
+
## 4. Marginal variable screening
|
|
120
|
+
|
|
121
|
+
Before committing to a multivariable model, screen each explanatory variable for a
|
|
122
|
+
marginal (unadjusted) signal against the outcome: fit a univariate logistic regression
|
|
123
|
+
per variable (or reuse the chi-square/t-test/Mann-Whitney results from step 3), and
|
|
124
|
+
rank by p-value or by AIC improvement over the null model.
|
|
125
|
+
|
|
126
|
+
This matters most when there are many candidate variables -- it narrows the field to a
|
|
127
|
+
manageable candidate set for the full model. Do not silently drop a variable just
|
|
128
|
+
because its bivariate test wasn't significant; a real effect can be masked by a
|
|
129
|
+
confounder and only emerge once other variables are adjusted for in step 5. Screening
|
|
130
|
+
informs the candidate list, it does not replace the full model.
|
|
131
|
+
|
|
132
|
+
Screening metrics and thresholds are summarized in `references/model_selection.md`.
|
|
133
|
+
|
|
134
|
+
## 5. Model selection and fitting, with explicit justification
|
|
135
|
+
|
|
136
|
+
Consult `references/model_selection.md` for the full decision table. The core logic:
|
|
137
|
+
|
|
138
|
+
- **Binary outcome -> logistic regression, not linear regression.** State explicitly
|
|
139
|
+
why: a linear model can predict outside [0, 1], and its error/variance structure
|
|
140
|
+
doesn't match a 0/1 target (violates linearity and homoscedasticity assumptions that
|
|
141
|
+
linear regression relies on). Continuous outcomes would call for linear regression
|
|
142
|
+
instead, and count outcomes for Poisson or negative binomial -- this decision logic
|
|
143
|
+
generalizes even though the immediate case here is binary.
|
|
144
|
+
- **Plain GLM vs. mixed-effects (GLMM)**: use a random effect for the clustering
|
|
145
|
+
variable identified in step 1 when observations are not independent (repeated
|
|
146
|
+
examples from the same model, prompt, or dataset). Use a plain GLM when observations
|
|
147
|
+
are reasonably independent. State this decision and the reasoning tied to the actual
|
|
148
|
+
data structure -- don't pick silently.
|
|
149
|
+
- **Candidate predictors**: start from the variables that passed step 4 screening, plus
|
|
150
|
+
any interaction flagged by the step 3 conditioning check.
|
|
151
|
+
- **Per-variable significance in the fitted model**: after fitting, test each
|
|
152
|
+
variable's significance with a Wald test or a likelihood-ratio test (comparing the
|
|
153
|
+
model with and without that term). Report this alongside the bivariate/screening
|
|
154
|
+
results from steps 3-4, so the reader can see whether a variable's marginal signal
|
|
155
|
+
holds up after adjusting for the others, or was actually explained by a confounder.
|
|
156
|
+
|
|
157
|
+
Use `scripts/fit_outcome_model.R` or `.py` as a starting template.
|
|
158
|
+
|
|
159
|
+
## 6. Goodness-of-fit and diagnostics
|
|
160
|
+
|
|
161
|
+
- Hosmer-Lemeshow test (or a suitable alternative when there are many continuous
|
|
162
|
+
predictors, since Hosmer-Lemeshow can be unreliable there).
|
|
163
|
+
- Deviance or Pearson residuals, binned residual plots.
|
|
164
|
+
- VIF for multicollinearity among the final model's predictors.
|
|
165
|
+
- Influence diagnostics (e.g. Cook's-distance analogs for GLMs).
|
|
166
|
+
- ROC curve and AUC for discrimination; a calibration plot for calibration.
|
|
167
|
+
- For GLMMs specifically: also check random-effect variance estimates, intraclass
|
|
168
|
+
correlation (ICC), and convergence warnings.
|
|
169
|
+
|
|
170
|
+
## 7. Visualize results
|
|
171
|
+
|
|
172
|
+
- Coefficient / odds-ratio forest plot with confidence intervals.
|
|
173
|
+
- Predicted-probability curves for the key continuous predictors (holding other
|
|
174
|
+
variables at a reference value or mean).
|
|
175
|
+
- ROC curve and calibration plot (carried over from step 6, presented as final results
|
|
176
|
+
rather than diagnostics here).
|
|
177
|
+
|
|
178
|
+
## 8. Conclusion and report
|
|
179
|
+
|
|
180
|
+
State which variables matter, in which direction, with what size and uncertainty, and
|
|
181
|
+
tie the finding back to the steps 2-5 evidence (distribution/outlier caveats, bivariate
|
|
182
|
+
and screening signal, adjusted-model significance) as corroboration or explanation.
|
|
183
|
+
|
|
184
|
+
Frame the conclusion around the specific research question recorded in step 1 -- not
|
|
185
|
+
as a generic statistical summary. Write it in plain language: the audience is
|
|
186
|
+
scientific researchers who understand research methodology but are not necessarily
|
|
187
|
+
statisticians, so translate statistical results into substantive meaning (for example,
|
|
188
|
+
"cases where X exceeded N were far more likely to fail" rather than reporting only an
|
|
189
|
+
odds ratio and a p-value), while still surfacing the effect size, uncertainty, and any
|
|
190
|
+
caveats a careful reader would need (small samples, assumption violations, correlated
|
|
191
|
+
predictors).
|
|
192
|
+
|
|
193
|
+
Figure styling conventions come from the installed chart-style/polish skills
|
|
194
|
+
(e.g. `eval-chart-style` for chart types and palette, `nature-figure` for
|
|
195
|
+
publication polish) -- do not duplicate those conventions here.
|
|
196
|
+
|
|
197
|
+
Save reproducibility information -- environment/package versions and any random seeds
|
|
198
|
+
used -- to `session_info.txt` alongside the analysis.
|
|
199
|
+
|
|
200
|
+
Start each new analysis under `projects/<analysis-name>/`, with the raw dataset copied
|
|
201
|
+
(read-only) into `data/`, outputs written to `output/figures/` and `output/tables/`,
|
|
202
|
+
and the writeup as `report.md` following `assets/analysis_report_template.md`.
|
|
203
|
+
|
|
204
|
+
## Reference files
|
|
205
|
+
|
|
206
|
+
- `references/model_selection.md` -- decision tables for outcome type x clustering x
|
|
207
|
+
assumption status -> model family, the univariate-test decision table from step 3,
|
|
208
|
+
and the marginal-screening metrics from step 4. Read this when deciding on a test or
|
|
209
|
+
model.
|
|
210
|
+
- `scripts/explanatory_var_eda.R` / `.py` -- step 2 starting template.
|
|
211
|
+
- `scripts/univariate_eda.R` / `.py` -- steps 3-4 starting template.
|
|
212
|
+
- `scripts/fit_outcome_model.R` / `.py` -- steps 5-7 starting template.
|
|
213
|
+
- `assets/analysis_report_template.md` -- report skeleton for step 8.
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
<!--
|
|
2
|
+
Report skeleton for step 8 of the outcome-driver-analysis skill. Copy this into
|
|
3
|
+
projects/<analysis-name>/report.md and fill it in.
|
|
4
|
+
|
|
5
|
+
Figure/table formatting and prose style are covered by the clear-technical-writing
|
|
6
|
+
skill -- apply its conventions when writing the actual content, don't duplicate them
|
|
7
|
+
here.
|
|
8
|
+
-->
|
|
9
|
+
|
|
10
|
+
# [Analysis title]
|
|
11
|
+
|
|
12
|
+
## Research question
|
|
13
|
+
|
|
14
|
+
What decision or understanding is this analysis meant to support (from intake, step 1)?
|
|
15
|
+
|
|
16
|
+
## Data
|
|
17
|
+
|
|
18
|
+
Source, size, outcome definition, explanatory variables and their types, any known
|
|
19
|
+
data-quality caveats.
|
|
20
|
+
|
|
21
|
+
## Explanatory-variable summary
|
|
22
|
+
|
|
23
|
+
Distributions, outliers, missingness, notable correlations among explanatory variables
|
|
24
|
+
(step 2).
|
|
25
|
+
|
|
26
|
+
## Univariate findings
|
|
27
|
+
|
|
28
|
+
How each explanatory variable relates to the outcome on its own, and what changes when
|
|
29
|
+
conditioning on other variables (step 3). Marginal screening results (step 4).
|
|
30
|
+
|
|
31
|
+
## Model and justification
|
|
32
|
+
|
|
33
|
+
Which model was fit and why (outcome type, independence/clustering structure, step 5).
|
|
34
|
+
Per-variable significance after adjusting for other predictors.
|
|
35
|
+
|
|
36
|
+
## Diagnostics
|
|
37
|
+
|
|
38
|
+
Goodness-of-fit, residuals, multicollinearity, discrimination and calibration (step 6).
|
|
39
|
+
|
|
40
|
+
## Results
|
|
41
|
+
|
|
42
|
+
Coefficient/odds-ratio plot, predicted-probability curves, ROC and calibration plots
|
|
43
|
+
(step 7).
|
|
44
|
+
|
|
45
|
+
## Conclusions
|
|
46
|
+
|
|
47
|
+
Plain-language answer to the research question above: which variables matter, in what
|
|
48
|
+
direction, how strongly, and with what caveats. Written for a scientific-researcher
|
|
49
|
+
audience, not a statistics audience (step 8).
|
|
50
|
+
|
|
51
|
+
## Reproducibility
|
|
52
|
+
|
|
53
|
+
Environment/package versions and random seeds used (see `session_info.txt`).
|