skill-harness 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- skill_harness/__init__.py +3 -0
- skill_harness/__main__.py +6 -0
- skill_harness/ablation/__init__.py +15 -0
- skill_harness/ablation/confound.py +362 -0
- skill_harness/ablation/operator.py +198 -0
- skill_harness/ablation/reconciler.py +138 -0
- skill_harness/ablation/render.py +252 -0
- skill_harness/ablation/runner.py +1534 -0
- skill_harness/ablation/sizing.py +183 -0
- skill_harness/ablation/stopping.py +274 -0
- skill_harness/ablation/subject.py +740 -0
- skill_harness/aggregation/__init__.py +44 -0
- skill_harness/aggregation/engine.py +767 -0
- skill_harness/aggregation/errors.py +88 -0
- skill_harness/aggregation/fit.py +349 -0
- skill_harness/aggregation/report.py +265 -0
- skill_harness/aggregation/status.py +205 -0
- skill_harness/aggregation/verdict.py +247 -0
- skill_harness/audit/__init__.py +117 -0
- skill_harness/cli/__init__.py +0 -0
- skill_harness/cli/diff_report.py +112 -0
- skill_harness/cli/main.py +2508 -0
- skill_harness/extractor/__init__.py +34 -0
- skill_harness/extractor/claude.py +320 -0
- skill_harness/extractor/errors.py +26 -0
- skill_harness/extractor/models.py +125 -0
- skill_harness/extractor/parser.py +103 -0
- skill_harness/extractor/pipeline.py +149 -0
- skill_harness/oracles/__init__.py +16 -0
- skill_harness/oracles/calibration/__init__.py +0 -0
- skill_harness/oracles/calibration/command.py +576 -0
- skill_harness/oracles/calibration/cost_projection.py +246 -0
- skill_harness/oracles/calibration/jsonl_parser.py +133 -0
- skill_harness/oracles/calibration/length_regression.py +133 -0
- skill_harness/oracles/errors.py +22 -0
- skill_harness/oracles/tier1/__init__.py +31 -0
- skill_harness/oracles/tier1/citation_presence_per_flag.py +211 -0
- skill_harness/oracles/tier1/compliance_proxy.py +114 -0
- skill_harness/oracles/tier1/fixtures/hedge_wordlist.json +61 -0
- skill_harness/oracles/tier1/hedge_index.py +114 -0
- skill_harness/oracles/tier1/structure_score.py +80 -0
- skill_harness/oracles/tier1/verbosity.py +85 -0
- skill_harness/oracles/tier2/__init__.py +1 -0
- skill_harness/oracles/tier2/injection_guard.py +51 -0
- skill_harness/oracles/tier2/judge.py +482 -0
- skill_harness/preflight.py +277 -0
- skill_harness/py.typed +0 -0
- skill_harness/storage/__init__.py +43 -0
- skill_harness/storage/context.py +61 -0
- skill_harness/storage/dual_write.py +183 -0
- skill_harness/storage/errors.py +47 -0
- skill_harness/storage/migrations.py +312 -0
- skill_harness/storage/migrations_sql/README.md +57 -0
- skill_harness/storage/migrations_sql/evidence/0001_initial.sql +221 -0
- skill_harness/storage/migrations_sql/evidence/0002_runs_trigger_split.sql +31 -0
- skill_harness/storage/migrations_sql/evidence/0003_admissible_verdicts_view.sql +35 -0
- skill_harness/storage/migrations_sql/evidence/0200_calibration_event_extensions.sql +34 -0
- skill_harness/storage/migrations_sql/evidence/0300_track_d_ablation.sql +63 -0
- skill_harness/storage/migrations_sql/evidence/0400_freeze_provenance.sql +40 -0
- skill_harness/storage/migrations_sql/evidence/0401_stale_frozen_view.sql +49 -0
- skill_harness/storage/migrations_sql/evidence/0500_subject_harness_pin.sql +19 -0
- skill_harness/storage/migrations_sql/evidence/0501_screen_store.sql +86 -0
- skill_harness/storage/migrations_sql/runtime/0001_initial.sql +77 -0
- skill_harness/storage/migrations_sql/runtime/0002_schema_migrations_triggers.sql +21 -0
- skill_harness/storage/models.py +709 -0
- skill_harness/storage/recovery.py +158 -0
- skill_harness/storage/repositories/__init__.py +6 -0
- skill_harness/storage/repositories/evidence/__init__.py +8 -0
- skill_harness/storage/repositories/evidence/calibration_events.py +138 -0
- skill_harness/storage/repositories/evidence/clauses.py +96 -0
- skill_harness/storage/repositories/evidence/confound_events.py +95 -0
- skill_harness/storage/repositories/evidence/frozen_cases.py +233 -0
- skill_harness/storage/repositories/evidence/judges.py +80 -0
- skill_harness/storage/repositories/evidence/metric_versions.py +88 -0
- skill_harness/storage/repositories/evidence/oracle_verdicts.py +82 -0
- skill_harness/storage/repositories/evidence/runs.py +104 -0
- skill_harness/storage/repositories/evidence/samples.py +110 -0
- skill_harness/storage/repositories/evidence/screens.py +152 -0
- skill_harness/storage/repositories/evidence/skills.py +86 -0
- skill_harness/storage/repositories/runtime/__init__.py +9 -0
- skill_harness/storage/repositories/runtime/cost_ledger.py +93 -0
- skill_harness/storage/repositories/runtime/current_calibration.py +130 -0
- skill_harness/storage/repositories/runtime/run_budget.py +131 -0
- skill_harness/storage/repositories/runtime/run_progress.py +86 -0
- skill_harness/storage/repositories/runtime/skill_imports_staging.py +88 -0
- skill_harness/storage/transaction.py +46 -0
- skill_harness/subject/__init__.py +21 -0
- skill_harness/subject/ingest.py +550 -0
- skill_harness/subject/inspect_adapter.py +408 -0
- skill_harness/subject/pin.py +184 -0
- skill_harness/subject/screen_backfill.py +185 -0
- skill_harness/subject/screen_ingest.py +208 -0
- skill_harness-0.2.1.dist-info/METADATA +240 -0
- skill_harness-0.2.1.dist-info/RECORD +97 -0
- skill_harness-0.2.1.dist-info/WHEEL +4 -0
- skill_harness-0.2.1.dist-info/entry_points.txt +2 -0
- skill_harness-0.2.1.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
"""Ablation runner — Track D.
|
|
2
|
+
|
|
3
|
+
D.1 provides:
|
|
4
|
+
- AblationOperator: versioned matched-length neutral substitution (operator.py)
|
|
5
|
+
- ConditionRenderer: renders Full/Ablated_k/Null conditions (render.py)
|
|
6
|
+
|
|
7
|
+
D.2 provides:
|
|
8
|
+
- AblationRunner: deterministic orchestration engine (runner.py)
|
|
9
|
+
- SubjectClient: subject model wrapper (subject.py)
|
|
10
|
+
- BetaBinomialAccumulator, StoppingReason: sequential stopping (stopping.py)
|
|
11
|
+
- NullAccumulator, detect_confounds, ConfoundEvent: confound monitoring (confound.py)
|
|
12
|
+
- reconcile_run_cost: cost reconciler (reconciler.py)
|
|
13
|
+
|
|
14
|
+
D.3 will add: CLI wiring in src/skill_harness/cli/main.py
|
|
15
|
+
"""
|
|
@@ -0,0 +1,362 @@
|
|
|
1
|
+
"""Confound monitoring for ablation runs (A11, A45, A46, A47).
|
|
2
|
+
|
|
3
|
+
Design:
|
|
4
|
+
- Score every sample on ALL registered metric_library axes in-memory.
|
|
5
|
+
- sigma(Null) estimated per (run, axis) at write-time from accumulated Null scores.
|
|
6
|
+
- N_null >= 30 floor: if fewer Null samples, confound detection is disabled for that axis.
|
|
7
|
+
- k = 2.0: emit a confound_events row ONLY when |delta| > k * sigma_Null.
|
|
8
|
+
- Threshold-triggered only -- NO dense per-sample x axis table (A47).
|
|
9
|
+
- Uncalibrated Tier-2 axes excluded from sigma(Null).
|
|
10
|
+
- Write-side assertion: primary_clause_id == ablated_clause_id (A46).
|
|
11
|
+
|
|
12
|
+
delta_kind values (A11):
|
|
13
|
+
- 'confound_flagged': movement on a claimed-but-other-clause axis (taints primary verdict)
|
|
14
|
+
- 'observed_unclaimed_delta': movement on a non-claimed axis (audit only, never aggregated)
|
|
15
|
+
|
|
16
|
+
This module is PURELY in-memory computation -- storage writes happen in the runner via
|
|
17
|
+
insert_confound_event(). The confound monitor returns events for the runner to persist.
|
|
18
|
+
|
|
19
|
+
Per A45: confound stays two-table; exclusion is the read-time VIEW (admissible_verdicts),
|
|
20
|
+
never a verdict-row state change. The confound_events row drives the VIEW exclusion.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
import contextlib
|
|
26
|
+
import statistics
|
|
27
|
+
from collections.abc import Callable
|
|
28
|
+
from dataclasses import dataclass
|
|
29
|
+
from typing import Final
|
|
30
|
+
|
|
31
|
+
# ---------------------------------------------------------------------------
|
|
32
|
+
# Constants
|
|
33
|
+
# ---------------------------------------------------------------------------
|
|
34
|
+
|
|
35
|
+
N_NULL_FLOOR: Final[int] = 30
|
|
36
|
+
"""Minimum Null samples before sigma(Null) is reliable enough for confound detection."""
|
|
37
|
+
|
|
38
|
+
K_THRESHOLD: Final[float] = 2.0
|
|
39
|
+
"""Sigma multiplier for confound threshold (A47)."""
|
|
40
|
+
|
|
41
|
+
# Standard Tier-1 axis names (matches function names in tier1 modules)
|
|
42
|
+
AXIS_VERBOSITY: Final[str] = "verbosity"
|
|
43
|
+
AXIS_HEDGE_INDEX: Final[str] = "hedge_index"
|
|
44
|
+
AXIS_STRUCTURE_SCORE: Final[str] = "structure_score"
|
|
45
|
+
AXIS_COMPLIANCE_PROXY: Final[str] = "compliance_proxy"
|
|
46
|
+
AXIS_CITATION_PRESENCE_PER_FLAG: Final[str] = "citation_presence_per_flag"
|
|
47
|
+
|
|
48
|
+
# ---------------------------------------------------------------------------
|
|
49
|
+
# Data types
|
|
50
|
+
# ---------------------------------------------------------------------------
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
@dataclass(frozen=True)
|
|
54
|
+
class ConfoundEvent:
|
|
55
|
+
"""A detected confound event (threshold-triggered, A47).
|
|
56
|
+
|
|
57
|
+
Does NOT map 1:1 to a DB row -- the runner calls insert_confound_event().
|
|
58
|
+
This is the pure-computation output of the confound monitor.
|
|
59
|
+
"""
|
|
60
|
+
|
|
61
|
+
primary_clause_id: str
|
|
62
|
+
"""The ablated clause whose verdict is being evaluated (A46: == ablated_clause_id)."""
|
|
63
|
+
|
|
64
|
+
affected_clause_id: str | None
|
|
65
|
+
"""The clause that owns the axis that moved (None for unclaimed axes)."""
|
|
66
|
+
|
|
67
|
+
axis: str
|
|
68
|
+
"""The metric axis name."""
|
|
69
|
+
|
|
70
|
+
delta: float
|
|
71
|
+
"""Observed delta (Full score - Ablated_k score) for this axis."""
|
|
72
|
+
|
|
73
|
+
null_sigma: float
|
|
74
|
+
"""Estimated sigma(Null) for this axis at time of detection."""
|
|
75
|
+
|
|
76
|
+
k_threshold: float
|
|
77
|
+
"""The k multiplier used (default K_THRESHOLD = 2.0)."""
|
|
78
|
+
|
|
79
|
+
delta_kind: str
|
|
80
|
+
"""'confound_flagged' or 'observed_unclaimed_delta'."""
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
# ---------------------------------------------------------------------------
|
|
84
|
+
# Metric scorer registry (injected by runner, not module-level globals)
|
|
85
|
+
# ---------------------------------------------------------------------------
|
|
86
|
+
|
|
87
|
+
# Type alias for a metric scoring function
|
|
88
|
+
MetricFn = Callable[[str], float]
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def get_default_tier1_scorers() -> dict[str, MetricFn]:
|
|
92
|
+
"""Return the default set of Tier-1 metric scoring functions.
|
|
93
|
+
|
|
94
|
+
Only imports when called (avoids import-time side effects in tests).
|
|
95
|
+
These are the Tier-1 metrics available per A14/A33.
|
|
96
|
+
"""
|
|
97
|
+
from skill_harness.oracles.tier1.citation_presence_per_flag import (
|
|
98
|
+
compute_citation_presence_per_flag,
|
|
99
|
+
)
|
|
100
|
+
from skill_harness.oracles.tier1.compliance_proxy import compute_compliance_proxy
|
|
101
|
+
from skill_harness.oracles.tier1.hedge_index import compute_hedge_index
|
|
102
|
+
from skill_harness.oracles.tier1.structure_score import compute_structure_score
|
|
103
|
+
from skill_harness.oracles.tier1.verbosity import count_tokens
|
|
104
|
+
|
|
105
|
+
return {
|
|
106
|
+
AXIS_VERBOSITY: count_tokens,
|
|
107
|
+
AXIS_HEDGE_INDEX: compute_hedge_index,
|
|
108
|
+
AXIS_STRUCTURE_SCORE: compute_structure_score,
|
|
109
|
+
AXIS_COMPLIANCE_PROXY: compute_compliance_proxy,
|
|
110
|
+
AXIS_CITATION_PRESENCE_PER_FLAG: compute_citation_presence_per_flag,
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
# ---------------------------------------------------------------------------
|
|
115
|
+
# Score computation
|
|
116
|
+
# ---------------------------------------------------------------------------
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def score_text_on_axes(text: str, scorers: dict[str, MetricFn]) -> dict[str, float]:
|
|
120
|
+
"""Score a text on all provided metric axes.
|
|
121
|
+
|
|
122
|
+
:param text: The output text to score.
|
|
123
|
+
:param scorers: Dict mapping axis_name -> scoring function.
|
|
124
|
+
:returns: Dict mapping axis_name -> float score.
|
|
125
|
+
"""
|
|
126
|
+
import contextlib
|
|
127
|
+
|
|
128
|
+
scores: dict[str, float] = {}
|
|
129
|
+
for axis, fn in scorers.items():
|
|
130
|
+
with contextlib.suppress(Exception):
|
|
131
|
+
scores[axis] = float(fn(text))
|
|
132
|
+
return scores
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
# ---------------------------------------------------------------------------
|
|
136
|
+
# NullAccumulator
|
|
137
|
+
# ---------------------------------------------------------------------------
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
class NullAccumulator:
|
|
141
|
+
"""Collects Null-condition scores for sigma(Null) estimation across all clauses.
|
|
142
|
+
|
|
143
|
+
The runner collects Null samples across ALL clauses in a run so that sigma(Null)
|
|
144
|
+
is estimated with sufficient power (N >= N_NULL_FLOOR per axis, A47).
|
|
145
|
+
"""
|
|
146
|
+
|
|
147
|
+
def __init__(
|
|
148
|
+
self,
|
|
149
|
+
scorers: dict[str, MetricFn] | None = None,
|
|
150
|
+
null_floor: int = N_NULL_FLOOR,
|
|
151
|
+
) -> None:
|
|
152
|
+
"""
|
|
153
|
+
:param scorers: Metric scoring functions (injected; defaults to Tier-1 set).
|
|
154
|
+
:param null_floor: Minimum Null samples per axis before sigma is reliable.
|
|
155
|
+
Defaults to ``N_NULL_FLOOR`` (30, A47). Lowered ONLY in tests for speed —
|
|
156
|
+
production code must use the default floor.
|
|
157
|
+
"""
|
|
158
|
+
self._scorers: dict[str, MetricFn] = (
|
|
159
|
+
scorers if scorers is not None else get_default_tier1_scorers()
|
|
160
|
+
)
|
|
161
|
+
self._null_floor: int = null_floor
|
|
162
|
+
self._axis_values: dict[str, list[float]] = {}
|
|
163
|
+
|
|
164
|
+
@property
|
|
165
|
+
def null_floor(self) -> int:
|
|
166
|
+
"""The configured Null-sample floor for this accumulator (A47)."""
|
|
167
|
+
return self._null_floor
|
|
168
|
+
|
|
169
|
+
def add(self, text: str) -> None:
|
|
170
|
+
"""Score text on all axes and accumulate for sigma estimation."""
|
|
171
|
+
scores = score_text_on_axes(text, self._scorers)
|
|
172
|
+
for axis, value in scores.items():
|
|
173
|
+
if axis not in self._axis_values:
|
|
174
|
+
self._axis_values[axis] = []
|
|
175
|
+
self._axis_values[axis].append(value)
|
|
176
|
+
|
|
177
|
+
def sigmas(self) -> dict[str, float]:
|
|
178
|
+
"""Compute sigma per axis. Returns only axes with >= null_floor samples.
|
|
179
|
+
|
|
180
|
+
A6: a floor-met, zero-variance axis is returned as sigma=0.0 rather than
|
|
181
|
+
dropped. Dropping it made "zero variance" and "below floor" both read as
|
|
182
|
+
"axis absent" to callers, which let confound detection go silently inert
|
|
183
|
+
for zero-variance axes -- see ``detect_confounds`` for the consuming side.
|
|
184
|
+
"""
|
|
185
|
+
result: dict[str, float] = {}
|
|
186
|
+
for axis, values in self._axis_values.items():
|
|
187
|
+
if len(values) < self._null_floor:
|
|
188
|
+
continue # Below floor -- confound detection disabled for this axis
|
|
189
|
+
with contextlib.suppress(statistics.StatisticsError):
|
|
190
|
+
result[axis] = statistics.stdev(values) # may be 0.0 (A6)
|
|
191
|
+
return result
|
|
192
|
+
|
|
193
|
+
def n(self) -> int:
|
|
194
|
+
"""Total Null sample count (max across axes)."""
|
|
195
|
+
if not self._axis_values:
|
|
196
|
+
return 0
|
|
197
|
+
return max(len(v) for v in self._axis_values.values())
|
|
198
|
+
|
|
199
|
+
def axis_n(self, axis: str) -> int:
|
|
200
|
+
"""Null sample count for a specific axis."""
|
|
201
|
+
return len(self._axis_values.get(axis, []))
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
# ---------------------------------------------------------------------------
|
|
205
|
+
# Confound detection (stateless -- uses pre-computed sigma)
|
|
206
|
+
# ---------------------------------------------------------------------------
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def detect_confounds(
|
|
210
|
+
full_text: str,
|
|
211
|
+
ablated_text: str,
|
|
212
|
+
null_sigmas: dict[str, float],
|
|
213
|
+
primary_clause_id: str,
|
|
214
|
+
clause_to_axis_map: dict[str, str],
|
|
215
|
+
scorers: dict[str, MetricFn] | None = None,
|
|
216
|
+
k: float = K_THRESHOLD,
|
|
217
|
+
) -> list[ConfoundEvent]:
|
|
218
|
+
"""Score Full vs Ablated_k on all axes and return confound events.
|
|
219
|
+
|
|
220
|
+
Stateless: uses pre-computed sigma(Null) from a NullAccumulator.
|
|
221
|
+
Threshold-triggered only -- no dense per-sample x axis table (A47).
|
|
222
|
+
|
|
223
|
+
:param full_text: Full-condition subject output.
|
|
224
|
+
:param ablated_text: Ablated_k-condition subject output.
|
|
225
|
+
:param null_sigmas: Pre-computed sigma(Null) per axis.
|
|
226
|
+
:param primary_clause_id: The ablated clause ID (A46 write-side assertion).
|
|
227
|
+
:param clause_to_axis_map: Maps clause_id -> claimed axis name.
|
|
228
|
+
:param scorers: Metric scoring functions (defaults to Tier-1 set).
|
|
229
|
+
:param k: Sigma multiplier.
|
|
230
|
+
:returns: List of ConfoundEvent (threshold-triggered only, A47).
|
|
231
|
+
"""
|
|
232
|
+
if scorers is None:
|
|
233
|
+
scorers = get_default_tier1_scorers()
|
|
234
|
+
|
|
235
|
+
full_scores = score_text_on_axes(full_text, scorers)
|
|
236
|
+
ablated_scores = score_text_on_axes(ablated_text, scorers)
|
|
237
|
+
|
|
238
|
+
# Invert clause_to_axis_map: axis -> clause_id (for delta classification)
|
|
239
|
+
axis_to_clause: dict[str, str] = {ax: cid for cid, ax in clause_to_axis_map.items()}
|
|
240
|
+
|
|
241
|
+
events: list[ConfoundEvent] = []
|
|
242
|
+
all_axes = set(full_scores.keys()) & set(ablated_scores.keys())
|
|
243
|
+
|
|
244
|
+
for axis in sorted(all_axes): # sorted for determinism
|
|
245
|
+
sigma = null_sigmas.get(axis)
|
|
246
|
+
if sigma is None:
|
|
247
|
+
continue # Below N_null floor -- genuinely insufficient Null data (A47)
|
|
248
|
+
|
|
249
|
+
delta = full_scores[axis] - ablated_scores[axis]
|
|
250
|
+
# A6: when sigma == 0.0 (floor met, zero variance under Null), k * sigma == 0,
|
|
251
|
+
# so this only screens out an EXACT-zero delta. Any nonzero delta falls
|
|
252
|
+
# through and is flagged -- a zero-variance Null gives no safe threshold to
|
|
253
|
+
# hide real movement behind, so confound monitoring must never go silently
|
|
254
|
+
# inert for a zero-variance axis.
|
|
255
|
+
if abs(delta) <= k * sigma:
|
|
256
|
+
continue # Within threshold -- not a confound event
|
|
257
|
+
|
|
258
|
+
# Determine delta_kind and affected_clause_id (A11)
|
|
259
|
+
owner_clause = axis_to_clause.get(axis)
|
|
260
|
+
|
|
261
|
+
if owner_clause is None:
|
|
262
|
+
# No clause claims this axis -> observed_unclaimed_delta (audit only)
|
|
263
|
+
events.append(
|
|
264
|
+
ConfoundEvent(
|
|
265
|
+
primary_clause_id=primary_clause_id,
|
|
266
|
+
affected_clause_id=None,
|
|
267
|
+
axis=axis,
|
|
268
|
+
delta=delta,
|
|
269
|
+
null_sigma=sigma,
|
|
270
|
+
k_threshold=k,
|
|
271
|
+
delta_kind="observed_unclaimed_delta",
|
|
272
|
+
)
|
|
273
|
+
)
|
|
274
|
+
elif owner_clause == primary_clause_id:
|
|
275
|
+
# Primary clause's own axis moved -- expected, not a confound
|
|
276
|
+
# (the runner's oracle measures this axis as the primary signal)
|
|
277
|
+
pass
|
|
278
|
+
else:
|
|
279
|
+
# Another clause's axis moved -> confound_flagged (taints primary verdict)
|
|
280
|
+
events.append(
|
|
281
|
+
ConfoundEvent(
|
|
282
|
+
primary_clause_id=primary_clause_id,
|
|
283
|
+
affected_clause_id=owner_clause,
|
|
284
|
+
axis=axis,
|
|
285
|
+
delta=delta,
|
|
286
|
+
null_sigma=sigma,
|
|
287
|
+
k_threshold=k,
|
|
288
|
+
delta_kind="confound_flagged",
|
|
289
|
+
)
|
|
290
|
+
)
|
|
291
|
+
|
|
292
|
+
return events
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
# ---------------------------------------------------------------------------
|
|
296
|
+
# Directional observation conversion
|
|
297
|
+
# ---------------------------------------------------------------------------
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
def delta_to_observation(
|
|
301
|
+
delta: float, comparator: str = "increase", tie_tolerance: float = 0.0
|
|
302
|
+
) -> float:
|
|
303
|
+
"""Convert a metric delta to a Win/Tie/Loss observation (A10 half-update).
|
|
304
|
+
|
|
305
|
+
:param delta: Full score - Ablated_k score on the primary axis.
|
|
306
|
+
:param comparator: The clause's persisted directional claim (schema CHECK:
|
|
307
|
+
increase|decrease|match). Determines which sign of delta counts as a Win:
|
|
308
|
+
- 'increase': Win = Full > Ablated (delta > 0). The clause claims the axis
|
|
309
|
+
rises -- this is the original, pre-A2 behavior and the default.
|
|
310
|
+
- 'decrease': Win = Full < Ablated (delta < 0). The clause claims the axis
|
|
311
|
+
falls, so the sign is flipped relative to 'increase' -- an effective
|
|
312
|
+
decrease-direction clause (e.g. "be concise" on verbosity) must verdict
|
|
313
|
+
Win, not Loss (A2).
|
|
314
|
+
- 'match'/anything else: treated as 'increase'. A "preserve" claim is a
|
|
315
|
+
magnitude-of-change question, not a delta-sign question, and defining that
|
|
316
|
+
comparison model is out of scope here -- this is a deliberate placeholder,
|
|
317
|
+
not a considered design decision.
|
|
318
|
+
:param tie_tolerance: Absolute tolerance for tie classification (0.0 = strict equality).
|
|
319
|
+
:returns: 1.0 (Win), 0.5 (Tie), 0.0 (Loss).
|
|
320
|
+
"""
|
|
321
|
+
if abs(delta) <= tie_tolerance:
|
|
322
|
+
return 0.5
|
|
323
|
+
if comparator == "decrease":
|
|
324
|
+
return 1.0 if delta < 0 else 0.0
|
|
325
|
+
return 1.0 if delta > 0 else 0.0
|
|
326
|
+
|
|
327
|
+
|
|
328
|
+
# ---------------------------------------------------------------------------
|
|
329
|
+
# QUAL-1: Length-confound detection for sub-tolerance clauses
|
|
330
|
+
# ---------------------------------------------------------------------------
|
|
331
|
+
|
|
332
|
+
|
|
333
|
+
def check_operator_length_tolerance(
|
|
334
|
+
clause_text: str,
|
|
335
|
+
placeholder_text: str,
|
|
336
|
+
tolerance_fraction: float = 0.10,
|
|
337
|
+
min_tolerance_tokens: int = 2,
|
|
338
|
+
) -> tuple[bool, int, int, int]:
|
|
339
|
+
"""Check whether the AblationOperator met its length tolerance for a clause.
|
|
340
|
+
|
|
341
|
+
QUAL-1 carry-forward from D.1: [ABLATED] is 4 tokens, so 1-2-token clauses
|
|
342
|
+
cannot achieve the +/-2-token minimum tolerance. The runner must detect this
|
|
343
|
+
and flag the Ablated_k samples as length-confounded (never emit clean-looking
|
|
344
|
+
verbosity evidence from them).
|
|
345
|
+
|
|
346
|
+
:param clause_text: Original clause text.
|
|
347
|
+
:param placeholder_text: AblationOperator placeholder for this clause.
|
|
348
|
+
:param tolerance_fraction: Token length tolerance fraction (default 10%).
|
|
349
|
+
:param min_tolerance_tokens: Minimum tolerance in tokens (default 2).
|
|
350
|
+
:returns: (within_tolerance, clause_tokens, placeholder_tokens, tolerance)
|
|
351
|
+
"""
|
|
352
|
+
# F4: delegate to the canonical cl100k_base counter (tier1/verbosity.py)
|
|
353
|
+
# instead of a third local tiktoken.get_encoding() re-implementation —
|
|
354
|
+
# this module already imports the same counter inside
|
|
355
|
+
# get_default_tier1_scorers(); a second copy here was the drift risk.
|
|
356
|
+
from skill_harness.oracles.tier1.verbosity import count_tokens
|
|
357
|
+
|
|
358
|
+
clause_tokens = count_tokens(clause_text)
|
|
359
|
+
placeholder_tokens = count_tokens(placeholder_text)
|
|
360
|
+
tolerance = max(min_tolerance_tokens, int(clause_tokens * tolerance_fraction))
|
|
361
|
+
within = abs(placeholder_tokens - clause_tokens) <= tolerance
|
|
362
|
+
return within, clause_tokens, placeholder_tokens, tolerance
|
|
@@ -0,0 +1,198 @@
|
|
|
1
|
+
"""AblationOperator — versioned matched-length neutral substitution (A39).
|
|
2
|
+
|
|
3
|
+
Design:
|
|
4
|
+
- Produces a **deterministic, byte-stable, matched-length, semantically-null**
|
|
5
|
+
placeholder for a removed clause. NOT deletion (A39: deletion conflates clause
|
|
6
|
+
axis-effect with prompt coherence / length / format loss).
|
|
7
|
+
- The placeholder token count approximates the removed clause's token count within
|
|
8
|
+
TOKEN_LENGTH_TOLERANCE_FRACTION (±10% or ±2 tokens, whichever is larger).
|
|
9
|
+
- ``implementation_hash`` is SHA-256 of the stable implementation-configuration
|
|
10
|
+
canonical string — mirrors the Tier-1 ``metric_versions.implementation_hash``
|
|
11
|
+
discipline so re-audit is possible if the operator changes version.
|
|
12
|
+
- ``ablation_operator_id`` is a stable version identifier string.
|
|
13
|
+
- NEVER calls a model — the deterministic Python layer constructs the placeholder.
|
|
14
|
+
- Uses ``tiktoken cl100k_base`` (version-pinned, offline) for token counting.
|
|
15
|
+
|
|
16
|
+
Algorithm:
|
|
17
|
+
1. Count tokens in the input clause text (offline tiktoken).
|
|
18
|
+
2. Divide the target token count by the filler unit token count to find how
|
|
19
|
+
many filler units to repeat.
|
|
20
|
+
3. Build the placeholder by repeating the filler unit, adding/removing partial
|
|
21
|
+
units until token count is within tolerance.
|
|
22
|
+
4. Return the placeholder string.
|
|
23
|
+
|
|
24
|
+
The filler unit is a semantically null, grammatically neutral phrase that:
|
|
25
|
+
- Has stable tokenization across tiktoken versions
|
|
26
|
+
- Produces no meaningful semantic signal to the subject model
|
|
27
|
+
- Has a known, stable token count (verified in ``FILLER_UNIT_TOKENS``)
|
|
28
|
+
|
|
29
|
+
Encoding determinism: tiktoken encoding is deterministic; PYTHONHASHSEED=0 is
|
|
30
|
+
not required but is enforced globally by the Tier-1 test suite.
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
from __future__ import annotations
|
|
34
|
+
|
|
35
|
+
import hashlib
|
|
36
|
+
from typing import TYPE_CHECKING, Final
|
|
37
|
+
|
|
38
|
+
from skill_harness.oracles.tier1.verbosity import ENCODING_NAME, get_encoding
|
|
39
|
+
|
|
40
|
+
if TYPE_CHECKING:
|
|
41
|
+
import tiktoken
|
|
42
|
+
|
|
43
|
+
# ---------------------------------------------------------------------------
|
|
44
|
+
# Constants
|
|
45
|
+
# ---------------------------------------------------------------------------
|
|
46
|
+
|
|
47
|
+
ABLATION_OPERATOR_VERSION: Final[str] = "neutral-filler-v1"
|
|
48
|
+
"""Stable version identifier for this operator implementation."""
|
|
49
|
+
|
|
50
|
+
# F-7 (S55 hostile review): ENCODING_NAME is imported from
|
|
51
|
+
# skill_harness.oracles.tier1.verbosity, the single source of truth for "which
|
|
52
|
+
# tiktoken encoding the harness uses" -- this module previously hand-duplicated
|
|
53
|
+
# its own "cl100k_base" literal (_ENCODING_NAME) and its own
|
|
54
|
+
# tiktoken.get_encoding() call, independent of verbosity.py's copy. A name
|
|
55
|
+
# change to the canonical encoding would have silently not propagated here.
|
|
56
|
+
|
|
57
|
+
# Semantically null filler phrase. Chosen to be:
|
|
58
|
+
# - Grammatically neutral (imperative-ish but content-free)
|
|
59
|
+
# - Stable tokenization in cl100k_base across the 0.7.x series
|
|
60
|
+
# - Not interpretable as an instruction to the subject model
|
|
61
|
+
_FILLER_UNIT: Final[str] = "[ABLATED]"
|
|
62
|
+
"""Base filler atom. Repeated to match target token count."""
|
|
63
|
+
|
|
64
|
+
# QUAL-2: pin FILLER_UNIT_TOKENS as a module-level constant so it is assertable by callers
|
|
65
|
+
# and not just private instance state. Computed once at module import time via the
|
|
66
|
+
# canonical encoding (F-7: get_encoding(), not a second tiktoken.get_encoding() call).
|
|
67
|
+
# This resolves the dangling docstring reference "verified in ``FILLER_UNIT_TOKENS``" (line 27).
|
|
68
|
+
FILLER_UNIT_TOKENS: Final[int] = len(get_encoding().encode(_FILLER_UNIT))
|
|
69
|
+
"""Stable token count of one _FILLER_UNIT atom in cl100k_base.
|
|
70
|
+
|
|
71
|
+
Pinned as a module constant (QUAL-2) so callers can assert invariants and the
|
|
72
|
+
docstring reference on line 27 ("verified in ``FILLER_UNIT_TOKENS``") resolves.
|
|
73
|
+
Change this constant whenever _FILLER_UNIT or the canonical encoding changes.
|
|
74
|
+
"""
|
|
75
|
+
|
|
76
|
+
# Tolerance for matched-length: ±10% of clause token count, minimum ±2 tokens.
|
|
77
|
+
TOKEN_LENGTH_TOLERANCE_FRACTION: Final[float] = 0.10
|
|
78
|
+
|
|
79
|
+
# Implementation configuration string — hashed to produce implementation_hash.
|
|
80
|
+
# Change this whenever the algorithm, filler text, or encoding changes.
|
|
81
|
+
_IMPL_CONFIG: Final[str] = (
|
|
82
|
+
f"ablation-operator version={ABLATION_OPERATOR_VERSION} "
|
|
83
|
+
f"filler={_FILLER_UNIT!r} "
|
|
84
|
+
f"encoding={ENCODING_NAME} "
|
|
85
|
+
f"tolerance={TOKEN_LENGTH_TOLERANCE_FRACTION}"
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
# ---------------------------------------------------------------------------
|
|
90
|
+
# AblationOperator
|
|
91
|
+
# ---------------------------------------------------------------------------
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
class AblationOperator:
|
|
95
|
+
"""Versioned matched-length neutral substitution operator.
|
|
96
|
+
|
|
97
|
+
Parameters
|
|
98
|
+
----------
|
|
99
|
+
None — all behaviour is fixed by the class constants above.
|
|
100
|
+
|
|
101
|
+
Usage
|
|
102
|
+
-----
|
|
103
|
+
>>> op = AblationOperator()
|
|
104
|
+
>>> placeholder = op.render("Always begin your response with 'Certainly!'.")
|
|
105
|
+
>>> op.ablation_operator_id
|
|
106
|
+
'neutral-filler-v1'
|
|
107
|
+
>>> len(op.implementation_hash)
|
|
108
|
+
64
|
|
109
|
+
"""
|
|
110
|
+
|
|
111
|
+
# ------------------------------------------------------------------
|
|
112
|
+
# Class-level stable attributes (shared across instances)
|
|
113
|
+
# ------------------------------------------------------------------
|
|
114
|
+
|
|
115
|
+
ablation_operator_id: str = ABLATION_OPERATOR_VERSION
|
|
116
|
+
implementation_hash: str = hashlib.sha256(_IMPL_CONFIG.encode("utf-8")).hexdigest()
|
|
117
|
+
|
|
118
|
+
def __init__(self) -> None:
|
|
119
|
+
# F-7: the canonical encoding (verbosity.get_encoding()), not a second
|
|
120
|
+
# independent tiktoken.get_encoding() call — see module docstring note.
|
|
121
|
+
self._enc: tiktoken.Encoding = get_encoding()
|
|
122
|
+
# Token count of one filler atom — pinned by FILLER_UNIT_TOKENS module constant (QUAL-2).
|
|
123
|
+
self._filler_unit_tokens: int = FILLER_UNIT_TOKENS
|
|
124
|
+
assert self._filler_unit_tokens > 0, (
|
|
125
|
+
f"FILLER_UNIT_TOKENS must be > 0; got {self._filler_unit_tokens!r} "
|
|
126
|
+
f"(QUAL-2: assertion on module constant)"
|
|
127
|
+
)
|
|
128
|
+
|
|
129
|
+
# ------------------------------------------------------------------
|
|
130
|
+
# Public API
|
|
131
|
+
# ------------------------------------------------------------------
|
|
132
|
+
|
|
133
|
+
def render(self, clause_text: str) -> str:
|
|
134
|
+
"""Return a deterministic, matched-length placeholder for ``clause_text``.
|
|
135
|
+
|
|
136
|
+
The returned placeholder:
|
|
137
|
+
- Is byte-equal across calls with the same input (deterministic).
|
|
138
|
+
- Has a token count within TOKEN_LENGTH_TOLERANCE_FRACTION of the
|
|
139
|
+
clause's token count (±10% or ±2 tokens, whichever is larger).
|
|
140
|
+
- Does NOT contain the original clause text.
|
|
141
|
+
- Is semantically null — conveys no axis-relevant signal to the model.
|
|
142
|
+
|
|
143
|
+
:param clause_text: The original clause text being ablated.
|
|
144
|
+
:returns: The placeholder string.
|
|
145
|
+
"""
|
|
146
|
+
target_tokens = len(self._enc.encode(clause_text))
|
|
147
|
+
return self._build_placeholder(target_tokens)
|
|
148
|
+
|
|
149
|
+
# ------------------------------------------------------------------
|
|
150
|
+
# Private helpers
|
|
151
|
+
# ------------------------------------------------------------------
|
|
152
|
+
|
|
153
|
+
def _build_placeholder(self, target_tokens: int) -> str:
|
|
154
|
+
"""Build a placeholder string of approximately ``target_tokens`` tokens.
|
|
155
|
+
|
|
156
|
+
Algorithm:
|
|
157
|
+
1. Compute how many whole filler units fit in target_tokens.
|
|
158
|
+
2. Build an initial placeholder.
|
|
159
|
+
3. Trim if over-length, pad if under-length — both by adjusting the
|
|
160
|
+
filler repetition count — until within tolerance.
|
|
161
|
+
|
|
162
|
+
The algorithm terminates because:
|
|
163
|
+
- We start close to the target (whole-unit count).
|
|
164
|
+
- Adjustment is monotone (add or remove one unit).
|
|
165
|
+
- Tolerance guarantees a valid range exists.
|
|
166
|
+
|
|
167
|
+
For zero-token clauses the placeholder is the single filler unit
|
|
168
|
+
(we never return an empty string — NOT deletion per A39).
|
|
169
|
+
"""
|
|
170
|
+
if target_tokens == 0:
|
|
171
|
+
return _FILLER_UNIT
|
|
172
|
+
|
|
173
|
+
filler_count = max(1, round(target_tokens / self._filler_unit_tokens))
|
|
174
|
+
|
|
175
|
+
# Separator between units is a space — stable token boundary
|
|
176
|
+
separator = " "
|
|
177
|
+
|
|
178
|
+
def _build(count: int) -> str:
|
|
179
|
+
return separator.join([_FILLER_UNIT] * count)
|
|
180
|
+
|
|
181
|
+
placeholder = _build(filler_count)
|
|
182
|
+
actual_tokens = len(self._enc.encode(placeholder))
|
|
183
|
+
|
|
184
|
+
tolerance = max(2, int(target_tokens * TOKEN_LENGTH_TOLERANCE_FRACTION))
|
|
185
|
+
|
|
186
|
+
# Adjust upward if under target
|
|
187
|
+
while actual_tokens < target_tokens - tolerance:
|
|
188
|
+
filler_count += 1
|
|
189
|
+
placeholder = _build(filler_count)
|
|
190
|
+
actual_tokens = len(self._enc.encode(placeholder))
|
|
191
|
+
|
|
192
|
+
# Adjust downward if over target
|
|
193
|
+
while actual_tokens > target_tokens + tolerance and filler_count > 1:
|
|
194
|
+
filler_count -= 1
|
|
195
|
+
placeholder = _build(filler_count)
|
|
196
|
+
actual_tokens = len(self._enc.encode(placeholder))
|
|
197
|
+
|
|
198
|
+
return placeholder
|