auditkit 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (81) hide show
  1. auditkit/README.md +99 -0
  2. auditkit/__init__.py +177 -0
  3. auditkit/__main__.py +3 -0
  4. auditkit/_bootstrap.py +77 -0
  5. auditkit/_identity_guard.py +99 -0
  6. auditkit/adapter.py +264 -0
  7. auditkit/annotator.py +339 -0
  8. auditkit/api.py +502 -0
  9. auditkit/assets/auditkit_logo.png +0 -0
  10. auditkit/cache.py +47 -0
  11. auditkit/cli.py +417 -0
  12. auditkit/comparison.py +563 -0
  13. auditkit/diff.py +265 -0
  14. auditkit/errors.py +54 -0
  15. auditkit/evaluator.py +20 -0
  16. auditkit/experiment.py +145 -0
  17. auditkit/hf_publish.py +262 -0
  18. auditkit/lmeval_engine.py +550 -0
  19. auditkit/loaders.py +121 -0
  20. auditkit/logs.py +18 -0
  21. auditkit/metric.py +199 -0
  22. auditkit/metrics/README.md +15 -0
  23. auditkit/metrics/__init__.py +0 -0
  24. auditkit/metrics/code.py +222 -0
  25. auditkit/metrics/embedding.py +131 -0
  26. auditkit/metrics/encoder_judge.py +423 -0
  27. auditkit/metrics/generation.py +331 -0
  28. auditkit/metrics/guard.py +412 -0
  29. auditkit/metrics/hallucination.py +45 -0
  30. auditkit/metrics/judge.py +547 -0
  31. auditkit/metrics/pairwise.py +153 -0
  32. auditkit/metrics/perf.py +53 -0
  33. auditkit/metrics/rag.py +149 -0
  34. auditkit/metrics/security.py +64 -0
  35. auditkit/metrics/toxicity.py +238 -0
  36. auditkit/model/README.md +16 -0
  37. auditkit/model/__init__.py +485 -0
  38. auditkit/model/anthropic.py +94 -0
  39. auditkit/model/api_gen.py +133 -0
  40. auditkit/model/groq_gen.py +121 -0
  41. auditkit/model/hf_gen.py +385 -0
  42. auditkit/model/lexsi.py +155 -0
  43. auditkit/model/litellm_gen.py +65 -0
  44. auditkit/model/openai.py +90 -0
  45. auditkit/model/openrouter_gen.py +152 -0
  46. auditkit/model/vllm_gen.py +316 -0
  47. auditkit/model_compare.py +655 -0
  48. auditkit/redteam/README.md +9 -0
  49. auditkit/redteam/__init__.py +26 -0
  50. auditkit/redteam/detector.py +37 -0
  51. auditkit/redteam/detectors/README.md +5 -0
  52. auditkit/redteam/detectors/builtin.py +126 -0
  53. auditkit/redteam/probe.py +39 -0
  54. auditkit/redteam/probes/README.md +5 -0
  55. auditkit/redteam/probes/builtin.py +85 -0
  56. auditkit/redteam/runner.py +206 -0
  57. auditkit/registry.py +65 -0
  58. auditkit/report.py +278 -0
  59. auditkit/report_format.py +52 -0
  60. auditkit/router.py +54 -0
  61. auditkit/runner.py +575 -0
  62. auditkit/runspec.py +159 -0
  63. auditkit/sample.py +40 -0
  64. auditkit/scenario.py +88 -0
  65. auditkit/scenarios/README.md +10 -0
  66. auditkit/scenarios/__init__.py +4 -0
  67. auditkit/scenarios/arc.py +33 -0
  68. auditkit/scenarios/gsm8k.py +32 -0
  69. auditkit/scenarios/hellaswag.py +33 -0
  70. auditkit/scenarios/humaneval.py +32 -0
  71. auditkit/scenarios/mmlu.py +34 -0
  72. auditkit/scenarios/truthfulqa.py +33 -0
  73. auditkit/score.py +165 -0
  74. auditkit/scorers.py +117 -0
  75. auditkit/scoring.py +79 -0
  76. auditkit/types.py +69 -0
  77. auditkit-1.0.0.dist-info/METADATA +396 -0
  78. auditkit-1.0.0.dist-info/RECORD +81 -0
  79. auditkit-1.0.0.dist-info/WHEEL +4 -0
  80. auditkit-1.0.0.dist-info/entry_points.txt +2 -0
  81. auditkit-1.0.0.dist-info/licenses/LICENSE.md +92 -0
auditkit/README.md ADDED
@@ -0,0 +1,99 @@
1
+ # AuditKIT — Source Layout
2
+
3
+ ```
4
+ auditkit/
5
+ ├── __init__.py # Public API exports, version, __all__
6
+ ├── __main__.py # python -m auditkit entry point
7
+ ├── adapter.py # Adapter ABC + 7 implementations (generation, chat, fewshot, etc.)
8
+ ├── annotator.py # Annotator ABC for post-generation annotation
9
+ ├── api.py # evaluate(), evaluate_many(), generate(), compare(), run_lmeval()
10
+ ├── cache.py # DiskCache — fingerprint-keyed result cache
11
+ ├── cli.py # argparse CLI: eval, init, list, redteam, compare
12
+ ├── comparison.py # RunComparison, MetricDelta, TaskDelta, grade_delta
13
+ ├── diff.py # RunDiff, DeltaGrade — comparison between runs
14
+ ├── errors.py # AuditKitError, CapabilityError, ExtraNotInstalled, etc.
15
+ ├── evaluator.py # Evaluator ABC for technique plugins
16
+ ├── experiment.py # Experiment, ExperimentDB — run management
17
+ ├── lmeval_engine.py # run_benchmark() — lm-evaluation-harness engine (extra)
18
+ ├── loaders.py # load_csv, load_hf, load_croissant
19
+ ├── logs.py # configure_logging
20
+ ├── metric.py # Metric ABC + ExactMatch, QuasiExactMatch, Acc, AccNorm
21
+ ├── model_compare.py # CompareResult, compare_models() — multi-model comparison
22
+ ├── registry.py # ObjectSpec, Registry — SCENARIOS, ADAPTERS, MODELS, etc.
23
+ ├── report.py # Prediction, RunResult — evaluation output
24
+ ├── report_format.py # Report — markdown formatting
25
+ ├── router.py # route_adapter() — auto-picks an Adapter from sample shape
26
+ ├── runner.py # Runner — 5-stage evaluation pipeline
27
+ ├── runspec.py # RunConfig, RunSpec, fingerprint
28
+ ├── sample.py # Sample
29
+ ├── scenario.py # Scenario ABC, ListScenario, CallableScenario
30
+ ├── score.py # Score, Stat, pass_at_k, pass_hat_k
31
+ ├── scorers.py # FunctionScorer, ScorerMetric, @scorer decorator
32
+ ├── scoring.py # ScoreGate, WeightedSum
33
+ ├── types.py # Enums: TaskKind, ScoreKind, DataType, etc.
34
+ │
35
+ ├── metrics/ # metric families
36
+ │ ├── code.py # Contains, Equals, F1Score, IsJson, Levenshtein, etc.
37
+ │ ├── embedding.py # CosineSimilarity, TokenOverlap, BM25Similarity
38
+ │ ├── generation.py # Bleu, RogueL, ChrF, BertScore, Perplexity, WordErrorRate
39
+ │ ├── hallucination.py # FactualConsistency
40
+ │ ├── judge.py # JudgeMetric, LLMJudge, GEval, RubricItem, Factuality, ClosedQA, Relevance
41
+ │ ├── pairwise.py # WinRate, EloScore, PreferenceAccuracy
42
+ │ ├── perf.py # LatencyStats, Throughput
43
+ │ ├── rag.py # LexicalGroundedness, ContextCoverage, ContextOverlap, AnswerOverlap
44
+ │ ├── security.py # DefconGrade, KeywordDetector, ThreatCategory
45
+ │ └── toxicity.py # ToxicityScore, RepresentationSkew, HateSpeechScore
46
+ │
47
+ ├── model/ # Model backends
48
+ │ ├── __init__.py # Request, Generated, Result_, Model ABC, AutoModel, CallableModel, PrecomputedModel
49
+ │ ├── anthropic.py # Anthropic backend (extra)
50
+ │ ├── api_gen.py # Generic API backend (extra)
51
+ │ ├── groq_gen.py # Groq backend (extra)
52
+ │ ├── hf_gen.py # HuggingFace Transformers backend (extra)
53
+ │ ├── lexsi.py # Lexsi gateway backend (extra)
54
+ │ ├── litellm_gen.py # LiteLLM backend (extra)
55
+ │ ├── openai.py # OpenAI backend (extra)
56
+ │ └── vllm_gen.py # vLLM backend (extra)
57
+ │
58
+ ├── redteam/ # Red teaming
59
+ │ ├── __init__.py # Probe, Detector, RedTeamRunner, RedTeamResult exports
60
+ │ ├── probe.py # Probe ABC + ProbeResult
61
+ │ ├── detector.py # Detector ABC + DetectorResult
62
+ │ ├── runner.py # RedTeamRunner, RedTeamResult — orchestration
63
+ │ ├── probes/
64
+ │ │ └── builtin.py # PromptInjectionProbe, JailbreakProbe, EncodingProbe, RefusalProbe
65
+ │ └── detectors/
66
+ │ └── builtin.py # KeywordDetector, RefusalDetector, InjectionSuccessDetector, etc.
67
+ │
68
+ └── scenarios/ # Built-in benchmark datasets
69
+ ├── arc.py # AI2 Reasoning Challenge
70
+ ├── gsm8k.py # Grade School Math 8K
71
+ ├── hellaswag.py # HellaSwag
72
+ ├── humaneval.py # HumanEval
73
+ ├── mmlu.py # Massive Multitask Language Understanding
74
+ └── truthfulqa.py # TruthfulQA
75
+ ```
76
+
77
+ ## Architecture (the spine)
78
+
79
+ Every evaluation flows through the same 5-stage pipeline:
80
+
81
+ ```
82
+ Dataset → Adapter → Model → Metrics → RunResult
83
+ ```
84
+
85
+ 1. Build requests — `Adapter.adapt(sample, config) → list[Request]`
86
+ 2. Execute — `Model.generate(requests) → list[Result_]`
87
+ 3. Annotate — `Annotator.annotate(sample, results) → dict`
88
+ 4. Score — `Metric.score(sample, output, context) → Score`
89
+ 5. Aggregate — `Stat.add(value)` → per-metric stats
90
+
91
+ ## Extending
92
+
93
+ **Add a metric:** Create a class inheriting `Metric`, implement `score()`, export via `__init__.py`.
94
+
95
+ **Add a model backend:** Create a file in `model/`, inherit `Model`, implement `generate()`, add prefix to `AutoModel._T1_BACKENDS`.
96
+
97
+ **Add a probe:** Create a class inheriting `Probe`, implement `prompts()`, register in the probe registry.
98
+
99
+ **Add a detector:** Create a class inheriting `Detector`, implement `detect()`, register in the detector registry.
auditkit/__init__.py ADDED
@@ -0,0 +1,177 @@
1
+ """auditkit — evaluate any model on any dataset and any task."""
2
+
3
+ from __future__ import annotations
4
+
5
+ __version__ = "1.0.0"
6
+
7
+ from .api import run_lmeval, compare, evaluate, evaluate_many, generate, scorer
8
+ from .lmeval_engine import BenchmarkEvaluator, map_model_spec, run_benchmark
9
+ from .loaders import load_csv, load_hf, load_croissant
10
+ from .model_compare import CompareResult, compare_models
11
+ from .experiment import Experiment, ExperimentDB
12
+ from .cache import DiskCache
13
+ from .logs import configure_logging
14
+ from .diff import DeltaGrade, RunDiff
15
+ from .comparison import MetricDelta, RunComparison, TaskDelta, grade_delta
16
+ from .registry import ADAPTERS, EVALUATORS, METRICS, SCENARIOS
17
+ from . import scenarios # noqa: F401 -- triggers @SCENARIOS.register decorators
18
+ from .errors import AuditKitError, ExtraNotInstalled
19
+ from .evaluator import Evaluator
20
+ from .model import AutoModel
21
+ from .scenario import CallableScenario, ListScenario, Scenario
22
+ from .metrics.code import (
23
+ Contains, EndsWith, Equals, F1Score, IsJson, Levenshtein, Regex, StartsWith, WordCount,
24
+ )
25
+ from .metrics.embedding import BM25Similarity, CosineSimilarity, TokenOverlap
26
+ from .metrics.generation import (
27
+ BertScore, Bleu, ChrF, Perplexity, RogueL, WordErrorRate,
28
+ )
29
+ from .metrics.encoder_judge import (
30
+ EncoderJudge, FactualityEncoderJudge, SentimentEncoderJudge,
31
+ )
32
+ from .metrics.hallucination import FactualConsistency
33
+ from .metrics.judge import (
34
+ BiasJudge, ClosedQA, Factuality, GEval, JudgeMetric, LLMJudge, Relevance, RubricItem,
35
+ )
36
+ from .metrics.pairwise import EloScore, PreferenceAccuracy, WinRate
37
+ from .metrics.toxicity import HateSpeechScore, RepresentationSkew, ToxicityScore
38
+ from .metrics.guard import GuardJudge
39
+ from .metrics.perf import LatencyStats, Throughput
40
+ from .metrics.rag import AnswerOverlap, ContextCoverage, ContextOverlap, LexicalGroundedness
41
+ from .metrics.security import DefconGrade, KeywordDetector, ThreatCategory
42
+ from .adapter import (
43
+ Adapter, ChatAdapter, FewShotAdapter, GenerationAdapter, InstructionAdapter, MCQAdapter,
44
+ RAGAdapter, TemplateAdapter,
45
+ )
46
+ from .router import route_adapter
47
+ from .annotator import Annotator, RegexAnnotator, LLMAnnotator, ThinkingStripAnnotator
48
+ from .report import RunResult
49
+ from .runspec import RunConfig
50
+ from .report_format import Report
51
+ from .sample import Sample
52
+ from .score import Score, Stat, pass_at_k, pass_hat_k
53
+ from .scoring import ScoreGate, WeightedSum
54
+ from .types import (
55
+ Capability,
56
+ DataType,
57
+ Direction,
58
+ ScoreKind,
59
+ Source,
60
+ TaskKind,
61
+ )
62
+
63
+ #: Public alias of :class:`RunResult` — the object returned by :func:`evaluate`.
64
+ Result = RunResult
65
+
66
+ __all__ = [
67
+ "evaluate",
68
+ "evaluate_many",
69
+ "run_lmeval",
70
+ "generate",
71
+ "compare",
72
+ "scorer",
73
+ "BenchmarkEvaluator",
74
+ "run_benchmark",
75
+ "map_model_spec",
76
+ "load_csv",
77
+ "load_hf",
78
+ "load_croissant",
79
+ "Experiment",
80
+ "ExperimentDB",
81
+ "Sample",
82
+ "Score",
83
+ "Stat",
84
+ "Result",
85
+ "RunConfig",
86
+ "Adapter",
87
+ "GenerationAdapter",
88
+ "MCQAdapter",
89
+ "ChatAdapter",
90
+ "FewShotAdapter",
91
+ "InstructionAdapter",
92
+ "RAGAdapter",
93
+ "TemplateAdapter",
94
+ "route_adapter",
95
+ "AutoModel",
96
+ "Scenario",
97
+ "ListScenario",
98
+ "CallableScenario",
99
+ "ADAPTERS",
100
+ "METRICS",
101
+ "EVALUATORS",
102
+ "ScoreGate",
103
+ "WeightedSum",
104
+ "Equals",
105
+ "configure_logging",
106
+ "BertScore",
107
+ "Bleu",
108
+ "ChrF",
109
+ "Contains",
110
+ "Perplexity",
111
+ "StartsWith",
112
+ "EndsWith",
113
+ "Evaluator",
114
+ "Regex",
115
+ "Levenshtein",
116
+ "WordCount",
117
+ "IsJson",
118
+ "RogueL",
119
+ "WordErrorRate",
120
+ "F1Score",
121
+ "RunDiff",
122
+ "DeltaGrade",
123
+ "RunComparison",
124
+ "MetricDelta",
125
+ "TaskDelta",
126
+ "grade_delta",
127
+ "LexicalGroundedness",
128
+ "ContextCoverage",
129
+ "ContextOverlap",
130
+ "AnswerOverlap",
131
+ "JudgeMetric",
132
+ "RubricItem",
133
+ "GEval",
134
+ "LLMJudge",
135
+ "EncoderJudge",
136
+ "FactualityEncoderJudge",
137
+ "SentimentEncoderJudge",
138
+ "Factuality",
139
+ "ClosedQA",
140
+ "Relevance",
141
+ "BiasJudge",
142
+ "KeywordDetector",
143
+ "DefconGrade",
144
+ "DiskCache",
145
+ "ThreatCategory",
146
+ "LatencyStats",
147
+ "Throughput",
148
+ "Annotator",
149
+ "RegexAnnotator",
150
+ "LLMAnnotator",
151
+ "ThinkingStripAnnotator",
152
+ "Report",
153
+ "TaskKind",
154
+ "ScoreKind",
155
+ "DataType",
156
+ "Direction",
157
+ "Capability",
158
+ "Source",
159
+ "SCENARIOS",
160
+ "AuditKitError",
161
+ "ExtraNotInstalled",
162
+ "CosineSimilarity",
163
+ "TokenOverlap",
164
+ "BM25Similarity",
165
+ "FactualConsistency",
166
+ "ToxicityScore",
167
+ "RepresentationSkew",
168
+ "HateSpeechScore",
169
+ "GuardJudge",
170
+ "WinRate",
171
+ "EloScore",
172
+ "PreferenceAccuracy",
173
+ "pass_at_k",
174
+ "pass_hat_k",
175
+ "CompareResult",
176
+ "compare_models",
177
+ ]
auditkit/__main__.py ADDED
@@ -0,0 +1,3 @@
1
+ from .cli import main
2
+
3
+ main()
auditkit/_bootstrap.py ADDED
@@ -0,0 +1,77 @@
1
+ """Shared paired-bootstrap significance test.
2
+
3
+ Both the model-comparison path (`model_compare.py`) and the experiment path
4
+ (`experiment.py`) need the same test — a two-sided paired bootstrap over
5
+ per-sample (baseline, candidate) score pairs — so it lives here once instead of
6
+ being reimplemented (and drifting) in each. The old copies also aligned the two
7
+ runs' scores by list *position*, which silently mispaired samples whenever the
8
+ runs' prediction order diverged (partial failures, retries, caching); callers
9
+ here align **by sample_id** via :func:`score_pairs`.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import random
15
+ from typing import Any, Optional
16
+
17
+ from .report import RunResult
18
+
19
+
20
+ def score_pairs(
21
+ baseline: RunResult, candidate: RunResult, metric: Optional[str] = None
22
+ ) -> list[tuple[float, float]]:
23
+ """Per-sample ``(baseline_value, candidate_value)`` pairs, aligned by ``sample_id``.
24
+
25
+ ``metric=None`` uses each prediction's primary ``.score`` (works for every
26
+ engine, including lm-eval benchmark runs whose predictions don't carry a
27
+ per-metric ``metadata["scores"]`` breakdown). A metric name reads that named
28
+ score out of ``metadata["scores"]`` (native runs).
29
+ """
30
+ def by_id(run: RunResult) -> dict[str, float]:
31
+ out: dict[str, float] = {}
32
+ for p in run.predictions:
33
+ if metric is None:
34
+ if p.score is not None:
35
+ out[p.sample_id] = float(p.score)
36
+ else:
37
+ for s in p.metadata.get("scores", []):
38
+ if s.get("name") == metric and s.get("value") is not None:
39
+ out[p.sample_id] = float(s["value"])
40
+ break
41
+ return out
42
+
43
+ b, c = by_id(baseline), by_id(candidate)
44
+ return [(b[i], c[i]) for i in sorted(b.keys() & c.keys())]
45
+
46
+
47
+ def paired_bootstrap(
48
+ pairs: list[tuple[float, float]], *, n_resamples: int = 1000, seed: int = 42
49
+ ) -> dict[str, Any]:
50
+ """Two-sided paired bootstrap over ``(baseline, candidate)`` value pairs.
51
+
52
+ Returns ``{n, mean_baseline, mean_candidate, delta, p_value, significant}``
53
+ where ``delta = mean_candidate - mean_baseline`` (positive ⇒ candidate scored
54
+ higher). Fewer than 2 pairs can't be resampled meaningfully → ``{"error": ...}``.
55
+ """
56
+ n = len(pairs)
57
+ if n < 2:
58
+ return {"error": "need at least 2 aligned sample pairs", "n": n}
59
+ diffs = [c - b for b, c in pairs]
60
+ mean_baseline = sum(b for b, _ in pairs) / n
61
+ mean_candidate = sum(c for _, c in pairs) / n
62
+ observed = sum(diffs) / n
63
+ rng = random.Random(seed)
64
+ count = 0
65
+ for _ in range(n_resamples):
66
+ resample = [rng.choice(diffs) for _ in range(n)]
67
+ if abs(sum(resample) / n) >= abs(observed):
68
+ count += 1
69
+ p = (count + 1) / (n_resamples + 1)
70
+ return {
71
+ "n": n,
72
+ "mean_baseline": round(mean_baseline, 6),
73
+ "mean_candidate": round(mean_candidate, 6),
74
+ "delta": round(observed, 6),
75
+ "p_value": round(p, 6),
76
+ "significant": p < 0.05,
77
+ }
@@ -0,0 +1,99 @@
1
+ """Shared class-definition-time guard against forgotten ``identity()``
2
+ overrides, used by :class:`Annotator`, :class:`Adapter`, and :class:`Metric`.
3
+
4
+ The bug this catches: a custom subclass takes real constructor config but
5
+ never overrides the base ``identity()`` (which only ever returns
6
+ ``{"name": self.name}``) -- two differently-configured instances then
7
+ silently produce the same ``RunSpec.fingerprint()``, and ``Runner.run()``
8
+ can return a stale cached ``RunResult`` scored under a different config,
9
+ with no error anywhere. Warning at class-definition time (via
10
+ ``__init_subclass__``) catches this as soon as the offending class is
11
+ *defined* -- before anyone even instantiates it, let alone runs anything.
12
+
13
+ One real false-positive this file exists to avoid: several built-in metrics
14
+ (``Contains``, ``StartsWith``, ``KeywordDetector``, ...) never override
15
+ ``identity()`` either, but they're not buggy -- they assign a
16
+ parameter-derived string to ``self.name`` inside ``__init__``
17
+ (e.g. ``self.name = f"contains({substring})"``), which already makes the
18
+ base ``identity()`` (``{"name": self.name}``) parameter-aware. Blindly
19
+ warning on "no identity() override" alone would fire on every one of those.
20
+ Detecting this safely means checking whether ``__init__`` assigns
21
+ ``self.name`` *without* instantiating the class (constructing an arbitrary
22
+ user class as a side effect of merely defining it is unacceptable -- some
23
+ constructors have required args with no defaults, and even when they don't,
24
+ running arbitrary user `__init__` code at import time is not something a
25
+ library should ever do implicitly). AST inspection of the already-loaded
26
+ source is the only way to check this without executing anything.
27
+ """
28
+
29
+ from __future__ import annotations
30
+
31
+ import ast
32
+ import inspect
33
+ import textwrap
34
+ import warnings
35
+ from typing import Any
36
+
37
+
38
+ def _assigns_self_attr(init_func: Any, attr: str) -> bool:
39
+ """True iff *init_func*'s body contains ``self.<attr> = ...`` anywhere
40
+ (a plain assignment, or a property setter reached via ``self.<attr> =``
41
+ -- both look identical in the AST).
42
+
43
+ Best-effort: classes defined somewhere ``inspect.getsource`` can't reach
44
+ (a REPL, ``exec``, certain notebook cells) fail open (return True, i.e.
45
+ "assume it's fine") rather than risk a false-positive warning on code
46
+ this check has no reliable way to inspect.
47
+ """
48
+ try:
49
+ source = textwrap.dedent(inspect.getsource(init_func))
50
+ tree = ast.parse(source)
51
+ except (OSError, TypeError, SyntaxError):
52
+ return True
53
+ for node in ast.walk(tree):
54
+ if not isinstance(node, ast.Assign):
55
+ continue
56
+ for target in node.targets:
57
+ if (
58
+ isinstance(target, ast.Attribute)
59
+ and target.attr == attr
60
+ and isinstance(target.value, ast.Name)
61
+ and target.value.id == "self"
62
+ ):
63
+ return True
64
+ return False
65
+
66
+
67
+ def warn_if_identity_incomplete(
68
+ cls: type, base_cls: type, main_method: str, name_attr: str = "name",
69
+ ) -> None:
70
+ """Warn if *cls* (a fresh subclass of *base_cls*) takes real constructor
71
+ config but neither overrides ``identity()`` nor derives ``self.<name_attr>``
72
+ from it -- called from ``base_cls.__init_subclass__``. *name_attr* is
73
+ ``"name"`` for :class:`Metric`/:class:`Annotator`, ``"method"`` for
74
+ :class:`Adapter` (whichever attribute that base's ``identity()`` keys off).
75
+ """
76
+ if cls.identity is not base_cls.identity:
77
+ return # this class (or a parent) already overrides identity()
78
+ init = cls.__dict__.get("__init__")
79
+ if init is None:
80
+ return # no new __init__ defined here -- nothing new to guard
81
+ params = [
82
+ p for pname, p in inspect.signature(init).parameters.items()
83
+ if pname != "self" and p.kind not in (p.VAR_POSITIONAL, p.VAR_KEYWORD)
84
+ ]
85
+ if not params:
86
+ return
87
+ if _assigns_self_attr(init, name_attr):
88
+ return # self.<name_attr> is parameter-derived -- base identity() is already fine
89
+ warnings.warn(
90
+ f"{cls.__name__} defines __init__ parameters "
91
+ f"({', '.join(p.name for p in params)}) but does not override "
92
+ f"{base_cls.__name__}.identity() (and doesn't derive self.{name_attr} "
93
+ f"from them either). Differently-configured instances will silently "
94
+ f"produce the same RunSpec.fingerprint() and can reuse a stale "
95
+ f"cached RunResult scored under a different config. Add an "
96
+ f"identity() override that includes every constructor argument "
97
+ f"affecting {main_method}()'s output.",
98
+ stacklevel=3,
99
+ )