auditkit 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- auditkit/README.md +99 -0
- auditkit/__init__.py +177 -0
- auditkit/__main__.py +3 -0
- auditkit/_bootstrap.py +77 -0
- auditkit/_identity_guard.py +99 -0
- auditkit/adapter.py +264 -0
- auditkit/annotator.py +339 -0
- auditkit/api.py +502 -0
- auditkit/assets/auditkit_logo.png +0 -0
- auditkit/cache.py +47 -0
- auditkit/cli.py +417 -0
- auditkit/comparison.py +563 -0
- auditkit/diff.py +265 -0
- auditkit/errors.py +54 -0
- auditkit/evaluator.py +20 -0
- auditkit/experiment.py +145 -0
- auditkit/hf_publish.py +262 -0
- auditkit/lmeval_engine.py +550 -0
- auditkit/loaders.py +121 -0
- auditkit/logs.py +18 -0
- auditkit/metric.py +199 -0
- auditkit/metrics/README.md +15 -0
- auditkit/metrics/__init__.py +0 -0
- auditkit/metrics/code.py +222 -0
- auditkit/metrics/embedding.py +131 -0
- auditkit/metrics/encoder_judge.py +423 -0
- auditkit/metrics/generation.py +331 -0
- auditkit/metrics/guard.py +412 -0
- auditkit/metrics/hallucination.py +45 -0
- auditkit/metrics/judge.py +547 -0
- auditkit/metrics/pairwise.py +153 -0
- auditkit/metrics/perf.py +53 -0
- auditkit/metrics/rag.py +149 -0
- auditkit/metrics/security.py +64 -0
- auditkit/metrics/toxicity.py +238 -0
- auditkit/model/README.md +16 -0
- auditkit/model/__init__.py +485 -0
- auditkit/model/anthropic.py +94 -0
- auditkit/model/api_gen.py +133 -0
- auditkit/model/groq_gen.py +121 -0
- auditkit/model/hf_gen.py +385 -0
- auditkit/model/lexsi.py +155 -0
- auditkit/model/litellm_gen.py +65 -0
- auditkit/model/openai.py +90 -0
- auditkit/model/openrouter_gen.py +152 -0
- auditkit/model/vllm_gen.py +316 -0
- auditkit/model_compare.py +655 -0
- auditkit/redteam/README.md +9 -0
- auditkit/redteam/__init__.py +26 -0
- auditkit/redteam/detector.py +37 -0
- auditkit/redteam/detectors/README.md +5 -0
- auditkit/redteam/detectors/builtin.py +126 -0
- auditkit/redteam/probe.py +39 -0
- auditkit/redteam/probes/README.md +5 -0
- auditkit/redteam/probes/builtin.py +85 -0
- auditkit/redteam/runner.py +206 -0
- auditkit/registry.py +65 -0
- auditkit/report.py +278 -0
- auditkit/report_format.py +52 -0
- auditkit/router.py +54 -0
- auditkit/runner.py +575 -0
- auditkit/runspec.py +159 -0
- auditkit/sample.py +40 -0
- auditkit/scenario.py +88 -0
- auditkit/scenarios/README.md +10 -0
- auditkit/scenarios/__init__.py +4 -0
- auditkit/scenarios/arc.py +33 -0
- auditkit/scenarios/gsm8k.py +32 -0
- auditkit/scenarios/hellaswag.py +33 -0
- auditkit/scenarios/humaneval.py +32 -0
- auditkit/scenarios/mmlu.py +34 -0
- auditkit/scenarios/truthfulqa.py +33 -0
- auditkit/score.py +165 -0
- auditkit/scorers.py +117 -0
- auditkit/scoring.py +79 -0
- auditkit/types.py +69 -0
- auditkit-1.0.0.dist-info/METADATA +396 -0
- auditkit-1.0.0.dist-info/RECORD +81 -0
- auditkit-1.0.0.dist-info/WHEEL +4 -0
- auditkit-1.0.0.dist-info/entry_points.txt +2 -0
- auditkit-1.0.0.dist-info/licenses/LICENSE.md +92 -0
auditkit/README.md
ADDED
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
# AuditKIT — Source Layout
|
|
2
|
+
|
|
3
|
+
```
|
|
4
|
+
auditkit/
|
|
5
|
+
├── __init__.py # Public API exports, version, __all__
|
|
6
|
+
├── __main__.py # python -m auditkit entry point
|
|
7
|
+
├── adapter.py # Adapter ABC + 7 implementations (generation, chat, fewshot, etc.)
|
|
8
|
+
├── annotator.py # Annotator ABC for post-generation annotation
|
|
9
|
+
├── api.py # evaluate(), evaluate_many(), generate(), compare(), run_lmeval()
|
|
10
|
+
├── cache.py # DiskCache — fingerprint-keyed result cache
|
|
11
|
+
├── cli.py # argparse CLI: eval, init, list, redteam, compare
|
|
12
|
+
├── comparison.py # RunComparison, MetricDelta, TaskDelta, grade_delta
|
|
13
|
+
├── diff.py # RunDiff, DeltaGrade — comparison between runs
|
|
14
|
+
├── errors.py # AuditKitError, CapabilityError, ExtraNotInstalled, etc.
|
|
15
|
+
├── evaluator.py # Evaluator ABC for technique plugins
|
|
16
|
+
├── experiment.py # Experiment, ExperimentDB — run management
|
|
17
|
+
├── lmeval_engine.py # run_benchmark() — lm-evaluation-harness engine (extra)
|
|
18
|
+
├── loaders.py # load_csv, load_hf, load_croissant
|
|
19
|
+
├── logs.py # configure_logging
|
|
20
|
+
├── metric.py # Metric ABC + ExactMatch, QuasiExactMatch, Acc, AccNorm
|
|
21
|
+
├── model_compare.py # CompareResult, compare_models() — multi-model comparison
|
|
22
|
+
├── registry.py # ObjectSpec, Registry — SCENARIOS, ADAPTERS, MODELS, etc.
|
|
23
|
+
├── report.py # Prediction, RunResult — evaluation output
|
|
24
|
+
├── report_format.py # Report — markdown formatting
|
|
25
|
+
├── router.py # route_adapter() — auto-picks an Adapter from sample shape
|
|
26
|
+
├── runner.py # Runner — 5-stage evaluation pipeline
|
|
27
|
+
├── runspec.py # RunConfig, RunSpec, fingerprint
|
|
28
|
+
├── sample.py # Sample
|
|
29
|
+
├── scenario.py # Scenario ABC, ListScenario, CallableScenario
|
|
30
|
+
├── score.py # Score, Stat, pass_at_k, pass_hat_k
|
|
31
|
+
├── scorers.py # FunctionScorer, ScorerMetric, @scorer decorator
|
|
32
|
+
├── scoring.py # ScoreGate, WeightedSum
|
|
33
|
+
├── types.py # Enums: TaskKind, ScoreKind, DataType, etc.
|
|
34
|
+
│
|
|
35
|
+
├── metrics/ # metric families
|
|
36
|
+
│ ├── code.py # Contains, Equals, F1Score, IsJson, Levenshtein, etc.
|
|
37
|
+
│ ├── embedding.py # CosineSimilarity, TokenOverlap, BM25Similarity
|
|
38
|
+
│ ├── generation.py # Bleu, RogueL, ChrF, BertScore, Perplexity, WordErrorRate
|
|
39
|
+
│ ├── hallucination.py # FactualConsistency
|
|
40
|
+
│ ├── judge.py # JudgeMetric, LLMJudge, GEval, RubricItem, Factuality, ClosedQA, Relevance
|
|
41
|
+
│ ├── pairwise.py # WinRate, EloScore, PreferenceAccuracy
|
|
42
|
+
│ ├── perf.py # LatencyStats, Throughput
|
|
43
|
+
│ ├── rag.py # LexicalGroundedness, ContextCoverage, ContextOverlap, AnswerOverlap
|
|
44
|
+
│ ├── security.py # DefconGrade, KeywordDetector, ThreatCategory
|
|
45
|
+
│ └── toxicity.py # ToxicityScore, RepresentationSkew, HateSpeechScore
|
|
46
|
+
│
|
|
47
|
+
├── model/ # Model backends
|
|
48
|
+
│ ├── __init__.py # Request, Generated, Result_, Model ABC, AutoModel, CallableModel, PrecomputedModel
|
|
49
|
+
│ ├── anthropic.py # Anthropic backend (extra)
|
|
50
|
+
│ ├── api_gen.py # Generic API backend (extra)
|
|
51
|
+
│ ├── groq_gen.py # Groq backend (extra)
|
|
52
|
+
│ ├── hf_gen.py # HuggingFace Transformers backend (extra)
|
|
53
|
+
│ ├── lexsi.py # Lexsi gateway backend (extra)
|
|
54
|
+
│ ├── litellm_gen.py # LiteLLM backend (extra)
|
|
55
|
+
│ ├── openai.py # OpenAI backend (extra)
|
|
56
|
+
│ └── vllm_gen.py # vLLM backend (extra)
|
|
57
|
+
│
|
|
58
|
+
├── redteam/ # Red teaming
|
|
59
|
+
│ ├── __init__.py # Probe, Detector, RedTeamRunner, RedTeamResult exports
|
|
60
|
+
│ ├── probe.py # Probe ABC + ProbeResult
|
|
61
|
+
│ ├── detector.py # Detector ABC + DetectorResult
|
|
62
|
+
│ ├── runner.py # RedTeamRunner, RedTeamResult — orchestration
|
|
63
|
+
│ ├── probes/
|
|
64
|
+
│ │ └── builtin.py # PromptInjectionProbe, JailbreakProbe, EncodingProbe, RefusalProbe
|
|
65
|
+
│ └── detectors/
|
|
66
|
+
│ └── builtin.py # KeywordDetector, RefusalDetector, InjectionSuccessDetector, etc.
|
|
67
|
+
│
|
|
68
|
+
└── scenarios/ # Built-in benchmark datasets
|
|
69
|
+
├── arc.py # AI2 Reasoning Challenge
|
|
70
|
+
├── gsm8k.py # Grade School Math 8K
|
|
71
|
+
├── hellaswag.py # HellaSwag
|
|
72
|
+
├── humaneval.py # HumanEval
|
|
73
|
+
├── mmlu.py # Massive Multitask Language Understanding
|
|
74
|
+
└── truthfulqa.py # TruthfulQA
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
## Architecture (the spine)
|
|
78
|
+
|
|
79
|
+
Every evaluation flows through the same 5-stage pipeline:
|
|
80
|
+
|
|
81
|
+
```
|
|
82
|
+
Dataset → Adapter → Model → Metrics → RunResult
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
1. Build requests — `Adapter.adapt(sample, config) → list[Request]`
|
|
86
|
+
2. Execute — `Model.generate(requests) → list[Result_]`
|
|
87
|
+
3. Annotate — `Annotator.annotate(sample, results) → dict`
|
|
88
|
+
4. Score — `Metric.score(sample, output, context) → Score`
|
|
89
|
+
5. Aggregate — `Stat.add(value)` → per-metric stats
|
|
90
|
+
|
|
91
|
+
## Extending
|
|
92
|
+
|
|
93
|
+
**Add a metric:** Create a class inheriting `Metric`, implement `score()`, export via `__init__.py`.
|
|
94
|
+
|
|
95
|
+
**Add a model backend:** Create a file in `model/`, inherit `Model`, implement `generate()`, add prefix to `AutoModel._T1_BACKENDS`.
|
|
96
|
+
|
|
97
|
+
**Add a probe:** Create a class inheriting `Probe`, implement `prompts()`, register in the probe registry.
|
|
98
|
+
|
|
99
|
+
**Add a detector:** Create a class inheriting `Detector`, implement `detect()`, register in the detector registry.
|
auditkit/__init__.py
ADDED
|
@@ -0,0 +1,177 @@
|
|
|
1
|
+
"""auditkit — evaluate any model on any dataset and any task."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
__version__ = "1.0.0"
|
|
6
|
+
|
|
7
|
+
from .api import run_lmeval, compare, evaluate, evaluate_many, generate, scorer
|
|
8
|
+
from .lmeval_engine import BenchmarkEvaluator, map_model_spec, run_benchmark
|
|
9
|
+
from .loaders import load_csv, load_hf, load_croissant
|
|
10
|
+
from .model_compare import CompareResult, compare_models
|
|
11
|
+
from .experiment import Experiment, ExperimentDB
|
|
12
|
+
from .cache import DiskCache
|
|
13
|
+
from .logs import configure_logging
|
|
14
|
+
from .diff import DeltaGrade, RunDiff
|
|
15
|
+
from .comparison import MetricDelta, RunComparison, TaskDelta, grade_delta
|
|
16
|
+
from .registry import ADAPTERS, EVALUATORS, METRICS, SCENARIOS
|
|
17
|
+
from . import scenarios # noqa: F401 -- triggers @SCENARIOS.register decorators
|
|
18
|
+
from .errors import AuditKitError, ExtraNotInstalled
|
|
19
|
+
from .evaluator import Evaluator
|
|
20
|
+
from .model import AutoModel
|
|
21
|
+
from .scenario import CallableScenario, ListScenario, Scenario
|
|
22
|
+
from .metrics.code import (
|
|
23
|
+
Contains, EndsWith, Equals, F1Score, IsJson, Levenshtein, Regex, StartsWith, WordCount,
|
|
24
|
+
)
|
|
25
|
+
from .metrics.embedding import BM25Similarity, CosineSimilarity, TokenOverlap
|
|
26
|
+
from .metrics.generation import (
|
|
27
|
+
BertScore, Bleu, ChrF, Perplexity, RogueL, WordErrorRate,
|
|
28
|
+
)
|
|
29
|
+
from .metrics.encoder_judge import (
|
|
30
|
+
EncoderJudge, FactualityEncoderJudge, SentimentEncoderJudge,
|
|
31
|
+
)
|
|
32
|
+
from .metrics.hallucination import FactualConsistency
|
|
33
|
+
from .metrics.judge import (
|
|
34
|
+
BiasJudge, ClosedQA, Factuality, GEval, JudgeMetric, LLMJudge, Relevance, RubricItem,
|
|
35
|
+
)
|
|
36
|
+
from .metrics.pairwise import EloScore, PreferenceAccuracy, WinRate
|
|
37
|
+
from .metrics.toxicity import HateSpeechScore, RepresentationSkew, ToxicityScore
|
|
38
|
+
from .metrics.guard import GuardJudge
|
|
39
|
+
from .metrics.perf import LatencyStats, Throughput
|
|
40
|
+
from .metrics.rag import AnswerOverlap, ContextCoverage, ContextOverlap, LexicalGroundedness
|
|
41
|
+
from .metrics.security import DefconGrade, KeywordDetector, ThreatCategory
|
|
42
|
+
from .adapter import (
|
|
43
|
+
Adapter, ChatAdapter, FewShotAdapter, GenerationAdapter, InstructionAdapter, MCQAdapter,
|
|
44
|
+
RAGAdapter, TemplateAdapter,
|
|
45
|
+
)
|
|
46
|
+
from .router import route_adapter
|
|
47
|
+
from .annotator import Annotator, RegexAnnotator, LLMAnnotator, ThinkingStripAnnotator
|
|
48
|
+
from .report import RunResult
|
|
49
|
+
from .runspec import RunConfig
|
|
50
|
+
from .report_format import Report
|
|
51
|
+
from .sample import Sample
|
|
52
|
+
from .score import Score, Stat, pass_at_k, pass_hat_k
|
|
53
|
+
from .scoring import ScoreGate, WeightedSum
|
|
54
|
+
from .types import (
|
|
55
|
+
Capability,
|
|
56
|
+
DataType,
|
|
57
|
+
Direction,
|
|
58
|
+
ScoreKind,
|
|
59
|
+
Source,
|
|
60
|
+
TaskKind,
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
#: Public alias of :class:`RunResult` — the object returned by :func:`evaluate`.
|
|
64
|
+
Result = RunResult
|
|
65
|
+
|
|
66
|
+
__all__ = [
|
|
67
|
+
"evaluate",
|
|
68
|
+
"evaluate_many",
|
|
69
|
+
"run_lmeval",
|
|
70
|
+
"generate",
|
|
71
|
+
"compare",
|
|
72
|
+
"scorer",
|
|
73
|
+
"BenchmarkEvaluator",
|
|
74
|
+
"run_benchmark",
|
|
75
|
+
"map_model_spec",
|
|
76
|
+
"load_csv",
|
|
77
|
+
"load_hf",
|
|
78
|
+
"load_croissant",
|
|
79
|
+
"Experiment",
|
|
80
|
+
"ExperimentDB",
|
|
81
|
+
"Sample",
|
|
82
|
+
"Score",
|
|
83
|
+
"Stat",
|
|
84
|
+
"Result",
|
|
85
|
+
"RunConfig",
|
|
86
|
+
"Adapter",
|
|
87
|
+
"GenerationAdapter",
|
|
88
|
+
"MCQAdapter",
|
|
89
|
+
"ChatAdapter",
|
|
90
|
+
"FewShotAdapter",
|
|
91
|
+
"InstructionAdapter",
|
|
92
|
+
"RAGAdapter",
|
|
93
|
+
"TemplateAdapter",
|
|
94
|
+
"route_adapter",
|
|
95
|
+
"AutoModel",
|
|
96
|
+
"Scenario",
|
|
97
|
+
"ListScenario",
|
|
98
|
+
"CallableScenario",
|
|
99
|
+
"ADAPTERS",
|
|
100
|
+
"METRICS",
|
|
101
|
+
"EVALUATORS",
|
|
102
|
+
"ScoreGate",
|
|
103
|
+
"WeightedSum",
|
|
104
|
+
"Equals",
|
|
105
|
+
"configure_logging",
|
|
106
|
+
"BertScore",
|
|
107
|
+
"Bleu",
|
|
108
|
+
"ChrF",
|
|
109
|
+
"Contains",
|
|
110
|
+
"Perplexity",
|
|
111
|
+
"StartsWith",
|
|
112
|
+
"EndsWith",
|
|
113
|
+
"Evaluator",
|
|
114
|
+
"Regex",
|
|
115
|
+
"Levenshtein",
|
|
116
|
+
"WordCount",
|
|
117
|
+
"IsJson",
|
|
118
|
+
"RogueL",
|
|
119
|
+
"WordErrorRate",
|
|
120
|
+
"F1Score",
|
|
121
|
+
"RunDiff",
|
|
122
|
+
"DeltaGrade",
|
|
123
|
+
"RunComparison",
|
|
124
|
+
"MetricDelta",
|
|
125
|
+
"TaskDelta",
|
|
126
|
+
"grade_delta",
|
|
127
|
+
"LexicalGroundedness",
|
|
128
|
+
"ContextCoverage",
|
|
129
|
+
"ContextOverlap",
|
|
130
|
+
"AnswerOverlap",
|
|
131
|
+
"JudgeMetric",
|
|
132
|
+
"RubricItem",
|
|
133
|
+
"GEval",
|
|
134
|
+
"LLMJudge",
|
|
135
|
+
"EncoderJudge",
|
|
136
|
+
"FactualityEncoderJudge",
|
|
137
|
+
"SentimentEncoderJudge",
|
|
138
|
+
"Factuality",
|
|
139
|
+
"ClosedQA",
|
|
140
|
+
"Relevance",
|
|
141
|
+
"BiasJudge",
|
|
142
|
+
"KeywordDetector",
|
|
143
|
+
"DefconGrade",
|
|
144
|
+
"DiskCache",
|
|
145
|
+
"ThreatCategory",
|
|
146
|
+
"LatencyStats",
|
|
147
|
+
"Throughput",
|
|
148
|
+
"Annotator",
|
|
149
|
+
"RegexAnnotator",
|
|
150
|
+
"LLMAnnotator",
|
|
151
|
+
"ThinkingStripAnnotator",
|
|
152
|
+
"Report",
|
|
153
|
+
"TaskKind",
|
|
154
|
+
"ScoreKind",
|
|
155
|
+
"DataType",
|
|
156
|
+
"Direction",
|
|
157
|
+
"Capability",
|
|
158
|
+
"Source",
|
|
159
|
+
"SCENARIOS",
|
|
160
|
+
"AuditKitError",
|
|
161
|
+
"ExtraNotInstalled",
|
|
162
|
+
"CosineSimilarity",
|
|
163
|
+
"TokenOverlap",
|
|
164
|
+
"BM25Similarity",
|
|
165
|
+
"FactualConsistency",
|
|
166
|
+
"ToxicityScore",
|
|
167
|
+
"RepresentationSkew",
|
|
168
|
+
"HateSpeechScore",
|
|
169
|
+
"GuardJudge",
|
|
170
|
+
"WinRate",
|
|
171
|
+
"EloScore",
|
|
172
|
+
"PreferenceAccuracy",
|
|
173
|
+
"pass_at_k",
|
|
174
|
+
"pass_hat_k",
|
|
175
|
+
"CompareResult",
|
|
176
|
+
"compare_models",
|
|
177
|
+
]
|
auditkit/__main__.py
ADDED
auditkit/_bootstrap.py
ADDED
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
"""Shared paired-bootstrap significance test.
|
|
2
|
+
|
|
3
|
+
Both the model-comparison path (`model_compare.py`) and the experiment path
|
|
4
|
+
(`experiment.py`) need the same test — a two-sided paired bootstrap over
|
|
5
|
+
per-sample (baseline, candidate) score pairs — so it lives here once instead of
|
|
6
|
+
being reimplemented (and drifting) in each. The old copies also aligned the two
|
|
7
|
+
runs' scores by list *position*, which silently mispaired samples whenever the
|
|
8
|
+
runs' prediction order diverged (partial failures, retries, caching); callers
|
|
9
|
+
here align **by sample_id** via :func:`score_pairs`.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import random
|
|
15
|
+
from typing import Any, Optional
|
|
16
|
+
|
|
17
|
+
from .report import RunResult
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def score_pairs(
|
|
21
|
+
baseline: RunResult, candidate: RunResult, metric: Optional[str] = None
|
|
22
|
+
) -> list[tuple[float, float]]:
|
|
23
|
+
"""Per-sample ``(baseline_value, candidate_value)`` pairs, aligned by ``sample_id``.
|
|
24
|
+
|
|
25
|
+
``metric=None`` uses each prediction's primary ``.score`` (works for every
|
|
26
|
+
engine, including lm-eval benchmark runs whose predictions don't carry a
|
|
27
|
+
per-metric ``metadata["scores"]`` breakdown). A metric name reads that named
|
|
28
|
+
score out of ``metadata["scores"]`` (native runs).
|
|
29
|
+
"""
|
|
30
|
+
def by_id(run: RunResult) -> dict[str, float]:
|
|
31
|
+
out: dict[str, float] = {}
|
|
32
|
+
for p in run.predictions:
|
|
33
|
+
if metric is None:
|
|
34
|
+
if p.score is not None:
|
|
35
|
+
out[p.sample_id] = float(p.score)
|
|
36
|
+
else:
|
|
37
|
+
for s in p.metadata.get("scores", []):
|
|
38
|
+
if s.get("name") == metric and s.get("value") is not None:
|
|
39
|
+
out[p.sample_id] = float(s["value"])
|
|
40
|
+
break
|
|
41
|
+
return out
|
|
42
|
+
|
|
43
|
+
b, c = by_id(baseline), by_id(candidate)
|
|
44
|
+
return [(b[i], c[i]) for i in sorted(b.keys() & c.keys())]
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def paired_bootstrap(
|
|
48
|
+
pairs: list[tuple[float, float]], *, n_resamples: int = 1000, seed: int = 42
|
|
49
|
+
) -> dict[str, Any]:
|
|
50
|
+
"""Two-sided paired bootstrap over ``(baseline, candidate)`` value pairs.
|
|
51
|
+
|
|
52
|
+
Returns ``{n, mean_baseline, mean_candidate, delta, p_value, significant}``
|
|
53
|
+
where ``delta = mean_candidate - mean_baseline`` (positive ⇒ candidate scored
|
|
54
|
+
higher). Fewer than 2 pairs can't be resampled meaningfully → ``{"error": ...}``.
|
|
55
|
+
"""
|
|
56
|
+
n = len(pairs)
|
|
57
|
+
if n < 2:
|
|
58
|
+
return {"error": "need at least 2 aligned sample pairs", "n": n}
|
|
59
|
+
diffs = [c - b for b, c in pairs]
|
|
60
|
+
mean_baseline = sum(b for b, _ in pairs) / n
|
|
61
|
+
mean_candidate = sum(c for _, c in pairs) / n
|
|
62
|
+
observed = sum(diffs) / n
|
|
63
|
+
rng = random.Random(seed)
|
|
64
|
+
count = 0
|
|
65
|
+
for _ in range(n_resamples):
|
|
66
|
+
resample = [rng.choice(diffs) for _ in range(n)]
|
|
67
|
+
if abs(sum(resample) / n) >= abs(observed):
|
|
68
|
+
count += 1
|
|
69
|
+
p = (count + 1) / (n_resamples + 1)
|
|
70
|
+
return {
|
|
71
|
+
"n": n,
|
|
72
|
+
"mean_baseline": round(mean_baseline, 6),
|
|
73
|
+
"mean_candidate": round(mean_candidate, 6),
|
|
74
|
+
"delta": round(observed, 6),
|
|
75
|
+
"p_value": round(p, 6),
|
|
76
|
+
"significant": p < 0.05,
|
|
77
|
+
}
|
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
"""Shared class-definition-time guard against forgotten ``identity()``
|
|
2
|
+
overrides, used by :class:`Annotator`, :class:`Adapter`, and :class:`Metric`.
|
|
3
|
+
|
|
4
|
+
The bug this catches: a custom subclass takes real constructor config but
|
|
5
|
+
never overrides the base ``identity()`` (which only ever returns
|
|
6
|
+
``{"name": self.name}``) -- two differently-configured instances then
|
|
7
|
+
silently produce the same ``RunSpec.fingerprint()``, and ``Runner.run()``
|
|
8
|
+
can return a stale cached ``RunResult`` scored under a different config,
|
|
9
|
+
with no error anywhere. Warning at class-definition time (via
|
|
10
|
+
``__init_subclass__``) catches this as soon as the offending class is
|
|
11
|
+
*defined* -- before anyone even instantiates it, let alone runs anything.
|
|
12
|
+
|
|
13
|
+
One real false-positive this file exists to avoid: several built-in metrics
|
|
14
|
+
(``Contains``, ``StartsWith``, ``KeywordDetector``, ...) never override
|
|
15
|
+
``identity()`` either, but they're not buggy -- they assign a
|
|
16
|
+
parameter-derived string to ``self.name`` inside ``__init__``
|
|
17
|
+
(e.g. ``self.name = f"contains({substring})"``), which already makes the
|
|
18
|
+
base ``identity()`` (``{"name": self.name}``) parameter-aware. Blindly
|
|
19
|
+
warning on "no identity() override" alone would fire on every one of those.
|
|
20
|
+
Detecting this safely means checking whether ``__init__`` assigns
|
|
21
|
+
``self.name`` *without* instantiating the class (constructing an arbitrary
|
|
22
|
+
user class as a side effect of merely defining it is unacceptable -- some
|
|
23
|
+
constructors have required args with no defaults, and even when they don't,
|
|
24
|
+
running arbitrary user `__init__` code at import time is not something a
|
|
25
|
+
library should ever do implicitly). AST inspection of the already-loaded
|
|
26
|
+
source is the only way to check this without executing anything.
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
from __future__ import annotations
|
|
30
|
+
|
|
31
|
+
import ast
|
|
32
|
+
import inspect
|
|
33
|
+
import textwrap
|
|
34
|
+
import warnings
|
|
35
|
+
from typing import Any
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _assigns_self_attr(init_func: Any, attr: str) -> bool:
|
|
39
|
+
"""True iff *init_func*'s body contains ``self.<attr> = ...`` anywhere
|
|
40
|
+
(a plain assignment, or a property setter reached via ``self.<attr> =``
|
|
41
|
+
-- both look identical in the AST).
|
|
42
|
+
|
|
43
|
+
Best-effort: classes defined somewhere ``inspect.getsource`` can't reach
|
|
44
|
+
(a REPL, ``exec``, certain notebook cells) fail open (return True, i.e.
|
|
45
|
+
"assume it's fine") rather than risk a false-positive warning on code
|
|
46
|
+
this check has no reliable way to inspect.
|
|
47
|
+
"""
|
|
48
|
+
try:
|
|
49
|
+
source = textwrap.dedent(inspect.getsource(init_func))
|
|
50
|
+
tree = ast.parse(source)
|
|
51
|
+
except (OSError, TypeError, SyntaxError):
|
|
52
|
+
return True
|
|
53
|
+
for node in ast.walk(tree):
|
|
54
|
+
if not isinstance(node, ast.Assign):
|
|
55
|
+
continue
|
|
56
|
+
for target in node.targets:
|
|
57
|
+
if (
|
|
58
|
+
isinstance(target, ast.Attribute)
|
|
59
|
+
and target.attr == attr
|
|
60
|
+
and isinstance(target.value, ast.Name)
|
|
61
|
+
and target.value.id == "self"
|
|
62
|
+
):
|
|
63
|
+
return True
|
|
64
|
+
return False
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def warn_if_identity_incomplete(
|
|
68
|
+
cls: type, base_cls: type, main_method: str, name_attr: str = "name",
|
|
69
|
+
) -> None:
|
|
70
|
+
"""Warn if *cls* (a fresh subclass of *base_cls*) takes real constructor
|
|
71
|
+
config but neither overrides ``identity()`` nor derives ``self.<name_attr>``
|
|
72
|
+
from it -- called from ``base_cls.__init_subclass__``. *name_attr* is
|
|
73
|
+
``"name"`` for :class:`Metric`/:class:`Annotator`, ``"method"`` for
|
|
74
|
+
:class:`Adapter` (whichever attribute that base's ``identity()`` keys off).
|
|
75
|
+
"""
|
|
76
|
+
if cls.identity is not base_cls.identity:
|
|
77
|
+
return # this class (or a parent) already overrides identity()
|
|
78
|
+
init = cls.__dict__.get("__init__")
|
|
79
|
+
if init is None:
|
|
80
|
+
return # no new __init__ defined here -- nothing new to guard
|
|
81
|
+
params = [
|
|
82
|
+
p for pname, p in inspect.signature(init).parameters.items()
|
|
83
|
+
if pname != "self" and p.kind not in (p.VAR_POSITIONAL, p.VAR_KEYWORD)
|
|
84
|
+
]
|
|
85
|
+
if not params:
|
|
86
|
+
return
|
|
87
|
+
if _assigns_self_attr(init, name_attr):
|
|
88
|
+
return # self.<name_attr> is parameter-derived -- base identity() is already fine
|
|
89
|
+
warnings.warn(
|
|
90
|
+
f"{cls.__name__} defines __init__ parameters "
|
|
91
|
+
f"({', '.join(p.name for p in params)}) but does not override "
|
|
92
|
+
f"{base_cls.__name__}.identity() (and doesn't derive self.{name_attr} "
|
|
93
|
+
f"from them either). Differently-configured instances will silently "
|
|
94
|
+
f"produce the same RunSpec.fingerprint() and can reuse a stale "
|
|
95
|
+
f"cached RunResult scored under a different config. Add an "
|
|
96
|
+
f"identity() override that includes every constructor argument "
|
|
97
|
+
f"affecting {main_method}()'s output.",
|
|
98
|
+
stacklevel=3,
|
|
99
|
+
)
|