skill-harness 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (97) hide show
  1. skill_harness/__init__.py +3 -0
  2. skill_harness/__main__.py +6 -0
  3. skill_harness/ablation/__init__.py +15 -0
  4. skill_harness/ablation/confound.py +362 -0
  5. skill_harness/ablation/operator.py +198 -0
  6. skill_harness/ablation/reconciler.py +138 -0
  7. skill_harness/ablation/render.py +252 -0
  8. skill_harness/ablation/runner.py +1534 -0
  9. skill_harness/ablation/sizing.py +183 -0
  10. skill_harness/ablation/stopping.py +274 -0
  11. skill_harness/ablation/subject.py +740 -0
  12. skill_harness/aggregation/__init__.py +44 -0
  13. skill_harness/aggregation/engine.py +767 -0
  14. skill_harness/aggregation/errors.py +88 -0
  15. skill_harness/aggregation/fit.py +349 -0
  16. skill_harness/aggregation/report.py +265 -0
  17. skill_harness/aggregation/status.py +205 -0
  18. skill_harness/aggregation/verdict.py +247 -0
  19. skill_harness/audit/__init__.py +117 -0
  20. skill_harness/cli/__init__.py +0 -0
  21. skill_harness/cli/diff_report.py +112 -0
  22. skill_harness/cli/main.py +2508 -0
  23. skill_harness/extractor/__init__.py +34 -0
  24. skill_harness/extractor/claude.py +320 -0
  25. skill_harness/extractor/errors.py +26 -0
  26. skill_harness/extractor/models.py +125 -0
  27. skill_harness/extractor/parser.py +103 -0
  28. skill_harness/extractor/pipeline.py +149 -0
  29. skill_harness/oracles/__init__.py +16 -0
  30. skill_harness/oracles/calibration/__init__.py +0 -0
  31. skill_harness/oracles/calibration/command.py +576 -0
  32. skill_harness/oracles/calibration/cost_projection.py +246 -0
  33. skill_harness/oracles/calibration/jsonl_parser.py +133 -0
  34. skill_harness/oracles/calibration/length_regression.py +133 -0
  35. skill_harness/oracles/errors.py +22 -0
  36. skill_harness/oracles/tier1/__init__.py +31 -0
  37. skill_harness/oracles/tier1/citation_presence_per_flag.py +211 -0
  38. skill_harness/oracles/tier1/compliance_proxy.py +114 -0
  39. skill_harness/oracles/tier1/fixtures/hedge_wordlist.json +61 -0
  40. skill_harness/oracles/tier1/hedge_index.py +114 -0
  41. skill_harness/oracles/tier1/structure_score.py +80 -0
  42. skill_harness/oracles/tier1/verbosity.py +85 -0
  43. skill_harness/oracles/tier2/__init__.py +1 -0
  44. skill_harness/oracles/tier2/injection_guard.py +51 -0
  45. skill_harness/oracles/tier2/judge.py +482 -0
  46. skill_harness/preflight.py +277 -0
  47. skill_harness/py.typed +0 -0
  48. skill_harness/storage/__init__.py +43 -0
  49. skill_harness/storage/context.py +61 -0
  50. skill_harness/storage/dual_write.py +183 -0
  51. skill_harness/storage/errors.py +47 -0
  52. skill_harness/storage/migrations.py +312 -0
  53. skill_harness/storage/migrations_sql/README.md +57 -0
  54. skill_harness/storage/migrations_sql/evidence/0001_initial.sql +221 -0
  55. skill_harness/storage/migrations_sql/evidence/0002_runs_trigger_split.sql +31 -0
  56. skill_harness/storage/migrations_sql/evidence/0003_admissible_verdicts_view.sql +35 -0
  57. skill_harness/storage/migrations_sql/evidence/0200_calibration_event_extensions.sql +34 -0
  58. skill_harness/storage/migrations_sql/evidence/0300_track_d_ablation.sql +63 -0
  59. skill_harness/storage/migrations_sql/evidence/0400_freeze_provenance.sql +40 -0
  60. skill_harness/storage/migrations_sql/evidence/0401_stale_frozen_view.sql +49 -0
  61. skill_harness/storage/migrations_sql/evidence/0500_subject_harness_pin.sql +19 -0
  62. skill_harness/storage/migrations_sql/evidence/0501_screen_store.sql +86 -0
  63. skill_harness/storage/migrations_sql/runtime/0001_initial.sql +77 -0
  64. skill_harness/storage/migrations_sql/runtime/0002_schema_migrations_triggers.sql +21 -0
  65. skill_harness/storage/models.py +709 -0
  66. skill_harness/storage/recovery.py +158 -0
  67. skill_harness/storage/repositories/__init__.py +6 -0
  68. skill_harness/storage/repositories/evidence/__init__.py +8 -0
  69. skill_harness/storage/repositories/evidence/calibration_events.py +138 -0
  70. skill_harness/storage/repositories/evidence/clauses.py +96 -0
  71. skill_harness/storage/repositories/evidence/confound_events.py +95 -0
  72. skill_harness/storage/repositories/evidence/frozen_cases.py +233 -0
  73. skill_harness/storage/repositories/evidence/judges.py +80 -0
  74. skill_harness/storage/repositories/evidence/metric_versions.py +88 -0
  75. skill_harness/storage/repositories/evidence/oracle_verdicts.py +82 -0
  76. skill_harness/storage/repositories/evidence/runs.py +104 -0
  77. skill_harness/storage/repositories/evidence/samples.py +110 -0
  78. skill_harness/storage/repositories/evidence/screens.py +152 -0
  79. skill_harness/storage/repositories/evidence/skills.py +86 -0
  80. skill_harness/storage/repositories/runtime/__init__.py +9 -0
  81. skill_harness/storage/repositories/runtime/cost_ledger.py +93 -0
  82. skill_harness/storage/repositories/runtime/current_calibration.py +130 -0
  83. skill_harness/storage/repositories/runtime/run_budget.py +131 -0
  84. skill_harness/storage/repositories/runtime/run_progress.py +86 -0
  85. skill_harness/storage/repositories/runtime/skill_imports_staging.py +88 -0
  86. skill_harness/storage/transaction.py +46 -0
  87. skill_harness/subject/__init__.py +21 -0
  88. skill_harness/subject/ingest.py +550 -0
  89. skill_harness/subject/inspect_adapter.py +408 -0
  90. skill_harness/subject/pin.py +184 -0
  91. skill_harness/subject/screen_backfill.py +185 -0
  92. skill_harness/subject/screen_ingest.py +208 -0
  93. skill_harness-0.2.1.dist-info/METADATA +240 -0
  94. skill_harness-0.2.1.dist-info/RECORD +97 -0
  95. skill_harness-0.2.1.dist-info/WHEEL +4 -0
  96. skill_harness-0.2.1.dist-info/entry_points.txt +2 -0
  97. skill_harness-0.2.1.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,3 @@
1
+ """Skill Harness — clause-ablation differential testing for LLM skills."""
2
+
3
+ __version__ = "0.2.0"
@@ -0,0 +1,6 @@
1
+ """Module entry point: ``python -m skill_harness`` == the ``skill-harness`` console script."""
2
+
3
+ from skill_harness.cli.main import cli
4
+
5
+ if __name__ == "__main__":
6
+ cli()
@@ -0,0 +1,15 @@
1
+ """Ablation runner — Track D.
2
+
3
+ D.1 provides:
4
+ - AblationOperator: versioned matched-length neutral substitution (operator.py)
5
+ - ConditionRenderer: renders Full/Ablated_k/Null conditions (render.py)
6
+
7
+ D.2 provides:
8
+ - AblationRunner: deterministic orchestration engine (runner.py)
9
+ - SubjectClient: subject model wrapper (subject.py)
10
+ - BetaBinomialAccumulator, StoppingReason: sequential stopping (stopping.py)
11
+ - NullAccumulator, detect_confounds, ConfoundEvent: confound monitoring (confound.py)
12
+ - reconcile_run_cost: cost reconciler (reconciler.py)
13
+
14
+ D.3 will add: CLI wiring in src/skill_harness/cli/main.py
15
+ """
@@ -0,0 +1,362 @@
1
+ """Confound monitoring for ablation runs (A11, A45, A46, A47).
2
+
3
+ Design:
4
+ - Score every sample on ALL registered metric_library axes in-memory.
5
+ - sigma(Null) estimated per (run, axis) at write-time from accumulated Null scores.
6
+ - N_null >= 30 floor: if fewer Null samples, confound detection is disabled for that axis.
7
+ - k = 2.0: emit a confound_events row ONLY when |delta| > k * sigma_Null.
8
+ - Threshold-triggered only -- NO dense per-sample x axis table (A47).
9
+ - Uncalibrated Tier-2 axes excluded from sigma(Null).
10
+ - Write-side assertion: primary_clause_id == ablated_clause_id (A46).
11
+
12
+ delta_kind values (A11):
13
+ - 'confound_flagged': movement on a claimed-but-other-clause axis (taints primary verdict)
14
+ - 'observed_unclaimed_delta': movement on a non-claimed axis (audit only, never aggregated)
15
+
16
+ This module is PURELY in-memory computation -- storage writes happen in the runner via
17
+ insert_confound_event(). The confound monitor returns events for the runner to persist.
18
+
19
+ Per A45: confound stays two-table; exclusion is the read-time VIEW (admissible_verdicts),
20
+ never a verdict-row state change. The confound_events row drives the VIEW exclusion.
21
+ """
22
+
23
+ from __future__ import annotations
24
+
25
+ import contextlib
26
+ import statistics
27
+ from collections.abc import Callable
28
+ from dataclasses import dataclass
29
+ from typing import Final
30
+
31
+ # ---------------------------------------------------------------------------
32
+ # Constants
33
+ # ---------------------------------------------------------------------------
34
+
35
+ N_NULL_FLOOR: Final[int] = 30
36
+ """Minimum Null samples before sigma(Null) is reliable enough for confound detection."""
37
+
38
+ K_THRESHOLD: Final[float] = 2.0
39
+ """Sigma multiplier for confound threshold (A47)."""
40
+
41
+ # Standard Tier-1 axis names (matches function names in tier1 modules)
42
+ AXIS_VERBOSITY: Final[str] = "verbosity"
43
+ AXIS_HEDGE_INDEX: Final[str] = "hedge_index"
44
+ AXIS_STRUCTURE_SCORE: Final[str] = "structure_score"
45
+ AXIS_COMPLIANCE_PROXY: Final[str] = "compliance_proxy"
46
+ AXIS_CITATION_PRESENCE_PER_FLAG: Final[str] = "citation_presence_per_flag"
47
+
48
+ # ---------------------------------------------------------------------------
49
+ # Data types
50
+ # ---------------------------------------------------------------------------
51
+
52
+
53
+ @dataclass(frozen=True)
54
+ class ConfoundEvent:
55
+ """A detected confound event (threshold-triggered, A47).
56
+
57
+ Does NOT map 1:1 to a DB row -- the runner calls insert_confound_event().
58
+ This is the pure-computation output of the confound monitor.
59
+ """
60
+
61
+ primary_clause_id: str
62
+ """The ablated clause whose verdict is being evaluated (A46: == ablated_clause_id)."""
63
+
64
+ affected_clause_id: str | None
65
+ """The clause that owns the axis that moved (None for unclaimed axes)."""
66
+
67
+ axis: str
68
+ """The metric axis name."""
69
+
70
+ delta: float
71
+ """Observed delta (Full score - Ablated_k score) for this axis."""
72
+
73
+ null_sigma: float
74
+ """Estimated sigma(Null) for this axis at time of detection."""
75
+
76
+ k_threshold: float
77
+ """The k multiplier used (default K_THRESHOLD = 2.0)."""
78
+
79
+ delta_kind: str
80
+ """'confound_flagged' or 'observed_unclaimed_delta'."""
81
+
82
+
83
+ # ---------------------------------------------------------------------------
84
+ # Metric scorer registry (injected by runner, not module-level globals)
85
+ # ---------------------------------------------------------------------------
86
+
87
+ # Type alias for a metric scoring function
88
+ MetricFn = Callable[[str], float]
89
+
90
+
91
+ def get_default_tier1_scorers() -> dict[str, MetricFn]:
92
+ """Return the default set of Tier-1 metric scoring functions.
93
+
94
+ Only imports when called (avoids import-time side effects in tests).
95
+ These are the Tier-1 metrics available per A14/A33.
96
+ """
97
+ from skill_harness.oracles.tier1.citation_presence_per_flag import (
98
+ compute_citation_presence_per_flag,
99
+ )
100
+ from skill_harness.oracles.tier1.compliance_proxy import compute_compliance_proxy
101
+ from skill_harness.oracles.tier1.hedge_index import compute_hedge_index
102
+ from skill_harness.oracles.tier1.structure_score import compute_structure_score
103
+ from skill_harness.oracles.tier1.verbosity import count_tokens
104
+
105
+ return {
106
+ AXIS_VERBOSITY: count_tokens,
107
+ AXIS_HEDGE_INDEX: compute_hedge_index,
108
+ AXIS_STRUCTURE_SCORE: compute_structure_score,
109
+ AXIS_COMPLIANCE_PROXY: compute_compliance_proxy,
110
+ AXIS_CITATION_PRESENCE_PER_FLAG: compute_citation_presence_per_flag,
111
+ }
112
+
113
+
114
+ # ---------------------------------------------------------------------------
115
+ # Score computation
116
+ # ---------------------------------------------------------------------------
117
+
118
+
119
+ def score_text_on_axes(text: str, scorers: dict[str, MetricFn]) -> dict[str, float]:
120
+ """Score a text on all provided metric axes.
121
+
122
+ :param text: The output text to score.
123
+ :param scorers: Dict mapping axis_name -> scoring function.
124
+ :returns: Dict mapping axis_name -> float score.
125
+ """
126
+ import contextlib
127
+
128
+ scores: dict[str, float] = {}
129
+ for axis, fn in scorers.items():
130
+ with contextlib.suppress(Exception):
131
+ scores[axis] = float(fn(text))
132
+ return scores
133
+
134
+
135
+ # ---------------------------------------------------------------------------
136
+ # NullAccumulator
137
+ # ---------------------------------------------------------------------------
138
+
139
+
140
+ class NullAccumulator:
141
+ """Collects Null-condition scores for sigma(Null) estimation across all clauses.
142
+
143
+ The runner collects Null samples across ALL clauses in a run so that sigma(Null)
144
+ is estimated with sufficient power (N >= N_NULL_FLOOR per axis, A47).
145
+ """
146
+
147
+ def __init__(
148
+ self,
149
+ scorers: dict[str, MetricFn] | None = None,
150
+ null_floor: int = N_NULL_FLOOR,
151
+ ) -> None:
152
+ """
153
+ :param scorers: Metric scoring functions (injected; defaults to Tier-1 set).
154
+ :param null_floor: Minimum Null samples per axis before sigma is reliable.
155
+ Defaults to ``N_NULL_FLOOR`` (30, A47). Lowered ONLY in tests for speed —
156
+ production code must use the default floor.
157
+ """
158
+ self._scorers: dict[str, MetricFn] = (
159
+ scorers if scorers is not None else get_default_tier1_scorers()
160
+ )
161
+ self._null_floor: int = null_floor
162
+ self._axis_values: dict[str, list[float]] = {}
163
+
164
+ @property
165
+ def null_floor(self) -> int:
166
+ """The configured Null-sample floor for this accumulator (A47)."""
167
+ return self._null_floor
168
+
169
+ def add(self, text: str) -> None:
170
+ """Score text on all axes and accumulate for sigma estimation."""
171
+ scores = score_text_on_axes(text, self._scorers)
172
+ for axis, value in scores.items():
173
+ if axis not in self._axis_values:
174
+ self._axis_values[axis] = []
175
+ self._axis_values[axis].append(value)
176
+
177
+ def sigmas(self) -> dict[str, float]:
178
+ """Compute sigma per axis. Returns only axes with >= null_floor samples.
179
+
180
+ A6: a floor-met, zero-variance axis is returned as sigma=0.0 rather than
181
+ dropped. Dropping it made "zero variance" and "below floor" both read as
182
+ "axis absent" to callers, which let confound detection go silently inert
183
+ for zero-variance axes -- see ``detect_confounds`` for the consuming side.
184
+ """
185
+ result: dict[str, float] = {}
186
+ for axis, values in self._axis_values.items():
187
+ if len(values) < self._null_floor:
188
+ continue # Below floor -- confound detection disabled for this axis
189
+ with contextlib.suppress(statistics.StatisticsError):
190
+ result[axis] = statistics.stdev(values) # may be 0.0 (A6)
191
+ return result
192
+
193
+ def n(self) -> int:
194
+ """Total Null sample count (max across axes)."""
195
+ if not self._axis_values:
196
+ return 0
197
+ return max(len(v) for v in self._axis_values.values())
198
+
199
+ def axis_n(self, axis: str) -> int:
200
+ """Null sample count for a specific axis."""
201
+ return len(self._axis_values.get(axis, []))
202
+
203
+
204
+ # ---------------------------------------------------------------------------
205
+ # Confound detection (stateless -- uses pre-computed sigma)
206
+ # ---------------------------------------------------------------------------
207
+
208
+
209
+ def detect_confounds(
210
+ full_text: str,
211
+ ablated_text: str,
212
+ null_sigmas: dict[str, float],
213
+ primary_clause_id: str,
214
+ clause_to_axis_map: dict[str, str],
215
+ scorers: dict[str, MetricFn] | None = None,
216
+ k: float = K_THRESHOLD,
217
+ ) -> list[ConfoundEvent]:
218
+ """Score Full vs Ablated_k on all axes and return confound events.
219
+
220
+ Stateless: uses pre-computed sigma(Null) from a NullAccumulator.
221
+ Threshold-triggered only -- no dense per-sample x axis table (A47).
222
+
223
+ :param full_text: Full-condition subject output.
224
+ :param ablated_text: Ablated_k-condition subject output.
225
+ :param null_sigmas: Pre-computed sigma(Null) per axis.
226
+ :param primary_clause_id: The ablated clause ID (A46 write-side assertion).
227
+ :param clause_to_axis_map: Maps clause_id -> claimed axis name.
228
+ :param scorers: Metric scoring functions (defaults to Tier-1 set).
229
+ :param k: Sigma multiplier.
230
+ :returns: List of ConfoundEvent (threshold-triggered only, A47).
231
+ """
232
+ if scorers is None:
233
+ scorers = get_default_tier1_scorers()
234
+
235
+ full_scores = score_text_on_axes(full_text, scorers)
236
+ ablated_scores = score_text_on_axes(ablated_text, scorers)
237
+
238
+ # Invert clause_to_axis_map: axis -> clause_id (for delta classification)
239
+ axis_to_clause: dict[str, str] = {ax: cid for cid, ax in clause_to_axis_map.items()}
240
+
241
+ events: list[ConfoundEvent] = []
242
+ all_axes = set(full_scores.keys()) & set(ablated_scores.keys())
243
+
244
+ for axis in sorted(all_axes): # sorted for determinism
245
+ sigma = null_sigmas.get(axis)
246
+ if sigma is None:
247
+ continue # Below N_null floor -- genuinely insufficient Null data (A47)
248
+
249
+ delta = full_scores[axis] - ablated_scores[axis]
250
+ # A6: when sigma == 0.0 (floor met, zero variance under Null), k * sigma == 0,
251
+ # so this only screens out an EXACT-zero delta. Any nonzero delta falls
252
+ # through and is flagged -- a zero-variance Null gives no safe threshold to
253
+ # hide real movement behind, so confound monitoring must never go silently
254
+ # inert for a zero-variance axis.
255
+ if abs(delta) <= k * sigma:
256
+ continue # Within threshold -- not a confound event
257
+
258
+ # Determine delta_kind and affected_clause_id (A11)
259
+ owner_clause = axis_to_clause.get(axis)
260
+
261
+ if owner_clause is None:
262
+ # No clause claims this axis -> observed_unclaimed_delta (audit only)
263
+ events.append(
264
+ ConfoundEvent(
265
+ primary_clause_id=primary_clause_id,
266
+ affected_clause_id=None,
267
+ axis=axis,
268
+ delta=delta,
269
+ null_sigma=sigma,
270
+ k_threshold=k,
271
+ delta_kind="observed_unclaimed_delta",
272
+ )
273
+ )
274
+ elif owner_clause == primary_clause_id:
275
+ # Primary clause's own axis moved -- expected, not a confound
276
+ # (the runner's oracle measures this axis as the primary signal)
277
+ pass
278
+ else:
279
+ # Another clause's axis moved -> confound_flagged (taints primary verdict)
280
+ events.append(
281
+ ConfoundEvent(
282
+ primary_clause_id=primary_clause_id,
283
+ affected_clause_id=owner_clause,
284
+ axis=axis,
285
+ delta=delta,
286
+ null_sigma=sigma,
287
+ k_threshold=k,
288
+ delta_kind="confound_flagged",
289
+ )
290
+ )
291
+
292
+ return events
293
+
294
+
295
+ # ---------------------------------------------------------------------------
296
+ # Directional observation conversion
297
+ # ---------------------------------------------------------------------------
298
+
299
+
300
+ def delta_to_observation(
301
+ delta: float, comparator: str = "increase", tie_tolerance: float = 0.0
302
+ ) -> float:
303
+ """Convert a metric delta to a Win/Tie/Loss observation (A10 half-update).
304
+
305
+ :param delta: Full score - Ablated_k score on the primary axis.
306
+ :param comparator: The clause's persisted directional claim (schema CHECK:
307
+ increase|decrease|match). Determines which sign of delta counts as a Win:
308
+ - 'increase': Win = Full > Ablated (delta > 0). The clause claims the axis
309
+ rises -- this is the original, pre-A2 behavior and the default.
310
+ - 'decrease': Win = Full < Ablated (delta < 0). The clause claims the axis
311
+ falls, so the sign is flipped relative to 'increase' -- an effective
312
+ decrease-direction clause (e.g. "be concise" on verbosity) must verdict
313
+ Win, not Loss (A2).
314
+ - 'match'/anything else: treated as 'increase'. A "preserve" claim is a
315
+ magnitude-of-change question, not a delta-sign question, and defining that
316
+ comparison model is out of scope here -- this is a deliberate placeholder,
317
+ not a considered design decision.
318
+ :param tie_tolerance: Absolute tolerance for tie classification (0.0 = strict equality).
319
+ :returns: 1.0 (Win), 0.5 (Tie), 0.0 (Loss).
320
+ """
321
+ if abs(delta) <= tie_tolerance:
322
+ return 0.5
323
+ if comparator == "decrease":
324
+ return 1.0 if delta < 0 else 0.0
325
+ return 1.0 if delta > 0 else 0.0
326
+
327
+
328
+ # ---------------------------------------------------------------------------
329
+ # QUAL-1: Length-confound detection for sub-tolerance clauses
330
+ # ---------------------------------------------------------------------------
331
+
332
+
333
+ def check_operator_length_tolerance(
334
+ clause_text: str,
335
+ placeholder_text: str,
336
+ tolerance_fraction: float = 0.10,
337
+ min_tolerance_tokens: int = 2,
338
+ ) -> tuple[bool, int, int, int]:
339
+ """Check whether the AblationOperator met its length tolerance for a clause.
340
+
341
+ QUAL-1 carry-forward from D.1: [ABLATED] is 4 tokens, so 1-2-token clauses
342
+ cannot achieve the +/-2-token minimum tolerance. The runner must detect this
343
+ and flag the Ablated_k samples as length-confounded (never emit clean-looking
344
+ verbosity evidence from them).
345
+
346
+ :param clause_text: Original clause text.
347
+ :param placeholder_text: AblationOperator placeholder for this clause.
348
+ :param tolerance_fraction: Token length tolerance fraction (default 10%).
349
+ :param min_tolerance_tokens: Minimum tolerance in tokens (default 2).
350
+ :returns: (within_tolerance, clause_tokens, placeholder_tokens, tolerance)
351
+ """
352
+ # F4: delegate to the canonical cl100k_base counter (tier1/verbosity.py)
353
+ # instead of a third local tiktoken.get_encoding() re-implementation —
354
+ # this module already imports the same counter inside
355
+ # get_default_tier1_scorers(); a second copy here was the drift risk.
356
+ from skill_harness.oracles.tier1.verbosity import count_tokens
357
+
358
+ clause_tokens = count_tokens(clause_text)
359
+ placeholder_tokens = count_tokens(placeholder_text)
360
+ tolerance = max(min_tolerance_tokens, int(clause_tokens * tolerance_fraction))
361
+ within = abs(placeholder_tokens - clause_tokens) <= tolerance
362
+ return within, clause_tokens, placeholder_tokens, tolerance
@@ -0,0 +1,198 @@
1
+ """AblationOperator — versioned matched-length neutral substitution (A39).
2
+
3
+ Design:
4
+ - Produces a **deterministic, byte-stable, matched-length, semantically-null**
5
+ placeholder for a removed clause. NOT deletion (A39: deletion conflates clause
6
+ axis-effect with prompt coherence / length / format loss).
7
+ - The placeholder token count approximates the removed clause's token count within
8
+ TOKEN_LENGTH_TOLERANCE_FRACTION (±10% or ±2 tokens, whichever is larger).
9
+ - ``implementation_hash`` is SHA-256 of the stable implementation-configuration
10
+ canonical string — mirrors the Tier-1 ``metric_versions.implementation_hash``
11
+ discipline so re-audit is possible if the operator changes version.
12
+ - ``ablation_operator_id`` is a stable version identifier string.
13
+ - NEVER calls a model — the deterministic Python layer constructs the placeholder.
14
+ - Uses ``tiktoken cl100k_base`` (version-pinned, offline) for token counting.
15
+
16
+ Algorithm:
17
+ 1. Count tokens in the input clause text (offline tiktoken).
18
+ 2. Divide the target token count by the filler unit token count to find how
19
+ many filler units to repeat.
20
+ 3. Build the placeholder by repeating the filler unit, adding/removing partial
21
+ units until token count is within tolerance.
22
+ 4. Return the placeholder string.
23
+
24
+ The filler unit is a semantically null, grammatically neutral phrase that:
25
+ - Has stable tokenization across tiktoken versions
26
+ - Produces no meaningful semantic signal to the subject model
27
+ - Has a known, stable token count (verified in ``FILLER_UNIT_TOKENS``)
28
+
29
+ Encoding determinism: tiktoken encoding is deterministic; PYTHONHASHSEED=0 is
30
+ not required but is enforced globally by the Tier-1 test suite.
31
+ """
32
+
33
+ from __future__ import annotations
34
+
35
+ import hashlib
36
+ from typing import TYPE_CHECKING, Final
37
+
38
+ from skill_harness.oracles.tier1.verbosity import ENCODING_NAME, get_encoding
39
+
40
+ if TYPE_CHECKING:
41
+ import tiktoken
42
+
43
+ # ---------------------------------------------------------------------------
44
+ # Constants
45
+ # ---------------------------------------------------------------------------
46
+
47
+ ABLATION_OPERATOR_VERSION: Final[str] = "neutral-filler-v1"
48
+ """Stable version identifier for this operator implementation."""
49
+
50
+ # F-7 (S55 hostile review): ENCODING_NAME is imported from
51
+ # skill_harness.oracles.tier1.verbosity, the single source of truth for "which
52
+ # tiktoken encoding the harness uses" -- this module previously hand-duplicated
53
+ # its own "cl100k_base" literal (_ENCODING_NAME) and its own
54
+ # tiktoken.get_encoding() call, independent of verbosity.py's copy. A name
55
+ # change to the canonical encoding would have silently not propagated here.
56
+
57
+ # Semantically null filler phrase. Chosen to be:
58
+ # - Grammatically neutral (imperative-ish but content-free)
59
+ # - Stable tokenization in cl100k_base across the 0.7.x series
60
+ # - Not interpretable as an instruction to the subject model
61
+ _FILLER_UNIT: Final[str] = "[ABLATED]"
62
+ """Base filler atom. Repeated to match target token count."""
63
+
64
+ # QUAL-2: pin FILLER_UNIT_TOKENS as a module-level constant so it is assertable by callers
65
+ # and not just private instance state. Computed once at module import time via the
66
+ # canonical encoding (F-7: get_encoding(), not a second tiktoken.get_encoding() call).
67
+ # This resolves the dangling docstring reference "verified in ``FILLER_UNIT_TOKENS``" (line 27).
68
+ FILLER_UNIT_TOKENS: Final[int] = len(get_encoding().encode(_FILLER_UNIT))
69
+ """Stable token count of one _FILLER_UNIT atom in cl100k_base.
70
+
71
+ Pinned as a module constant (QUAL-2) so callers can assert invariants and the
72
+ docstring reference on line 27 ("verified in ``FILLER_UNIT_TOKENS``") resolves.
73
+ Change this constant whenever _FILLER_UNIT or the canonical encoding changes.
74
+ """
75
+
76
+ # Tolerance for matched-length: ±10% of clause token count, minimum ±2 tokens.
77
+ TOKEN_LENGTH_TOLERANCE_FRACTION: Final[float] = 0.10
78
+
79
+ # Implementation configuration string — hashed to produce implementation_hash.
80
+ # Change this whenever the algorithm, filler text, or encoding changes.
81
+ _IMPL_CONFIG: Final[str] = (
82
+ f"ablation-operator version={ABLATION_OPERATOR_VERSION} "
83
+ f"filler={_FILLER_UNIT!r} "
84
+ f"encoding={ENCODING_NAME} "
85
+ f"tolerance={TOKEN_LENGTH_TOLERANCE_FRACTION}"
86
+ )
87
+
88
+
89
+ # ---------------------------------------------------------------------------
90
+ # AblationOperator
91
+ # ---------------------------------------------------------------------------
92
+
93
+
94
+ class AblationOperator:
95
+ """Versioned matched-length neutral substitution operator.
96
+
97
+ Parameters
98
+ ----------
99
+ None — all behaviour is fixed by the class constants above.
100
+
101
+ Usage
102
+ -----
103
+ >>> op = AblationOperator()
104
+ >>> placeholder = op.render("Always begin your response with 'Certainly!'.")
105
+ >>> op.ablation_operator_id
106
+ 'neutral-filler-v1'
107
+ >>> len(op.implementation_hash)
108
+ 64
109
+ """
110
+
111
+ # ------------------------------------------------------------------
112
+ # Class-level stable attributes (shared across instances)
113
+ # ------------------------------------------------------------------
114
+
115
+ ablation_operator_id: str = ABLATION_OPERATOR_VERSION
116
+ implementation_hash: str = hashlib.sha256(_IMPL_CONFIG.encode("utf-8")).hexdigest()
117
+
118
+ def __init__(self) -> None:
119
+ # F-7: the canonical encoding (verbosity.get_encoding()), not a second
120
+ # independent tiktoken.get_encoding() call — see module docstring note.
121
+ self._enc: tiktoken.Encoding = get_encoding()
122
+ # Token count of one filler atom — pinned by FILLER_UNIT_TOKENS module constant (QUAL-2).
123
+ self._filler_unit_tokens: int = FILLER_UNIT_TOKENS
124
+ assert self._filler_unit_tokens > 0, (
125
+ f"FILLER_UNIT_TOKENS must be > 0; got {self._filler_unit_tokens!r} "
126
+ f"(QUAL-2: assertion on module constant)"
127
+ )
128
+
129
+ # ------------------------------------------------------------------
130
+ # Public API
131
+ # ------------------------------------------------------------------
132
+
133
+ def render(self, clause_text: str) -> str:
134
+ """Return a deterministic, matched-length placeholder for ``clause_text``.
135
+
136
+ The returned placeholder:
137
+ - Is byte-equal across calls with the same input (deterministic).
138
+ - Has a token count within TOKEN_LENGTH_TOLERANCE_FRACTION of the
139
+ clause's token count (±10% or ±2 tokens, whichever is larger).
140
+ - Does NOT contain the original clause text.
141
+ - Is semantically null — conveys no axis-relevant signal to the model.
142
+
143
+ :param clause_text: The original clause text being ablated.
144
+ :returns: The placeholder string.
145
+ """
146
+ target_tokens = len(self._enc.encode(clause_text))
147
+ return self._build_placeholder(target_tokens)
148
+
149
+ # ------------------------------------------------------------------
150
+ # Private helpers
151
+ # ------------------------------------------------------------------
152
+
153
+ def _build_placeholder(self, target_tokens: int) -> str:
154
+ """Build a placeholder string of approximately ``target_tokens`` tokens.
155
+
156
+ Algorithm:
157
+ 1. Compute how many whole filler units fit in target_tokens.
158
+ 2. Build an initial placeholder.
159
+ 3. Trim if over-length, pad if under-length — both by adjusting the
160
+ filler repetition count — until within tolerance.
161
+
162
+ The algorithm terminates because:
163
+ - We start close to the target (whole-unit count).
164
+ - Adjustment is monotone (add or remove one unit).
165
+ - Tolerance guarantees a valid range exists.
166
+
167
+ For zero-token clauses the placeholder is the single filler unit
168
+ (we never return an empty string — NOT deletion per A39).
169
+ """
170
+ if target_tokens == 0:
171
+ return _FILLER_UNIT
172
+
173
+ filler_count = max(1, round(target_tokens / self._filler_unit_tokens))
174
+
175
+ # Separator between units is a space — stable token boundary
176
+ separator = " "
177
+
178
+ def _build(count: int) -> str:
179
+ return separator.join([_FILLER_UNIT] * count)
180
+
181
+ placeholder = _build(filler_count)
182
+ actual_tokens = len(self._enc.encode(placeholder))
183
+
184
+ tolerance = max(2, int(target_tokens * TOKEN_LENGTH_TOLERANCE_FRACTION))
185
+
186
+ # Adjust upward if under target
187
+ while actual_tokens < target_tokens - tolerance:
188
+ filler_count += 1
189
+ placeholder = _build(filler_count)
190
+ actual_tokens = len(self._enc.encode(placeholder))
191
+
192
+ # Adjust downward if over target
193
+ while actual_tokens > target_tokens + tolerance and filler_count > 1:
194
+ filler_count -= 1
195
+ placeholder = _build(filler_count)
196
+ actual_tokens = len(self._enc.encode(placeholder))
197
+
198
+ return placeholder