anchortest 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- anchortest-0.1.0/.github/FUNDING.yml +8 -0
- anchortest-0.1.0/.github/workflows/ci.yml +21 -0
- anchortest-0.1.0/.github/workflows/publish.yml +50 -0
- anchortest-0.1.0/.gitignore +8 -0
- anchortest-0.1.0/LICENSE +21 -0
- anchortest-0.1.0/MANIFESTO.md +110 -0
- anchortest-0.1.0/PKG-INFO +152 -0
- anchortest-0.1.0/README.md +130 -0
- anchortest-0.1.0/RELEASING.md +46 -0
- anchortest-0.1.0/docs/index.html +376 -0
- anchortest-0.1.0/docs/launch_posts.md +136 -0
- anchortest-0.1.0/examples/random_walk_example.py +59 -0
- anchortest-0.1.0/pyproject.toml +36 -0
- anchortest-0.1.0/src/anchortest/__init__.py +26 -0
- anchortest-0.1.0/src/anchortest/anchors.py +69 -0
- anchortest-0.1.0/src/anchortest/artifacts.py +87 -0
- anchortest-0.1.0/src/anchortest/compare.py +44 -0
- anchortest-0.1.0/src/anchortest/metrics.py +69 -0
- anchortest-0.1.0/src/anchortest/mutation.py +70 -0
- anchortest-0.1.0/tests/mutation_check.py +53 -0
- anchortest-0.1.0/tests/test_anchors.py +62 -0
- anchortest-0.1.0/tests/test_artifacts.py +54 -0
- anchortest-0.1.0/tests/test_compare.py +50 -0
- anchortest-0.1.0/tests/test_metrics.py +40 -0
- anchortest-0.1.0/tests/test_mutation.py +65 -0
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
# GitHub Sponsors / funding links for this repo.
|
|
2
|
+
# REPLACE-ME placeholders below before this takes effect -- GitHub ignores unset fields, but a literal
|
|
3
|
+
# "REPLACE-ME" left in place will show a broken link. Fill in only the ones you actually set up:
|
|
4
|
+
# - github: your GitHub username, after enabling GitHub Sponsors on your account (Settings > Sponsors)
|
|
5
|
+
# - custom: a list of URLs (e.g. a Gumroad or Stripe Payment Link page), if you set one up
|
|
6
|
+
|
|
7
|
+
github: REPLACE-ME
|
|
8
|
+
# custom: ["https://REPLACE-ME"]
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
name: CI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
pull_request:
|
|
6
|
+
|
|
7
|
+
jobs:
|
|
8
|
+
test:
|
|
9
|
+
runs-on: ubuntu-latest
|
|
10
|
+
strategy:
|
|
11
|
+
matrix:
|
|
12
|
+
python-version: ["3.9", "3.11", "3.12"]
|
|
13
|
+
steps:
|
|
14
|
+
- uses: actions/checkout@v4
|
|
15
|
+
- uses: actions/setup-python@v5
|
|
16
|
+
with:
|
|
17
|
+
python-version: ${{ matrix.python-version }}
|
|
18
|
+
- run: pip install -e ".[dev]"
|
|
19
|
+
- run: python -m unittest discover -s tests
|
|
20
|
+
- run: python tests/mutation_check.py
|
|
21
|
+
- run: python examples/random_walk_example.py
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
name: Publish to PyPI
|
|
2
|
+
|
|
3
|
+
# Runs when you publish a GitHub Release. Uses PyPI's Trusted Publishing (OIDC) -- no API token stored
|
|
4
|
+
# anywhere, in this repo or on PyPI. One-time setup required on PyPI's side; see RELEASING.md.
|
|
5
|
+
on:
|
|
6
|
+
release:
|
|
7
|
+
types: [published]
|
|
8
|
+
|
|
9
|
+
permissions:
|
|
10
|
+
contents: read
|
|
11
|
+
|
|
12
|
+
jobs:
|
|
13
|
+
test:
|
|
14
|
+
runs-on: ubuntu-latest
|
|
15
|
+
steps:
|
|
16
|
+
- uses: actions/checkout@v4
|
|
17
|
+
- uses: actions/setup-python@v5
|
|
18
|
+
with:
|
|
19
|
+
python-version: "3.11"
|
|
20
|
+
- run: pip install -e ".[dev]"
|
|
21
|
+
- run: python -m unittest discover -s tests
|
|
22
|
+
- run: python tests/mutation_check.py
|
|
23
|
+
|
|
24
|
+
build:
|
|
25
|
+
needs: test
|
|
26
|
+
runs-on: ubuntu-latest
|
|
27
|
+
steps:
|
|
28
|
+
- uses: actions/checkout@v4
|
|
29
|
+
- uses: actions/setup-python@v5
|
|
30
|
+
with:
|
|
31
|
+
python-version: "3.11"
|
|
32
|
+
- run: pip install build
|
|
33
|
+
- run: python -m build
|
|
34
|
+
- uses: actions/upload-artifact@v4
|
|
35
|
+
with:
|
|
36
|
+
name: dist
|
|
37
|
+
path: dist/
|
|
38
|
+
|
|
39
|
+
publish:
|
|
40
|
+
needs: build
|
|
41
|
+
runs-on: ubuntu-latest
|
|
42
|
+
environment: pypi
|
|
43
|
+
permissions:
|
|
44
|
+
id-token: write # required for OIDC trusted publishing -- this is the only credential involved
|
|
45
|
+
steps:
|
|
46
|
+
- uses: actions/download-artifact@v4
|
|
47
|
+
with:
|
|
48
|
+
name: dist
|
|
49
|
+
path: dist/
|
|
50
|
+
- uses: pypa/gh-action-pypi-publish@release/v1
|
anchortest-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Grant Miller
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
# Your backtest is lying to you, and it's not the bug you're looking for
|
|
2
|
+
|
|
3
|
+
Every backtester knows to check for lookahead bias. Fewer check for the three things below — because each
|
|
4
|
+
one produces a result that looks completely honest. The code runs. The number is real. The chart is
|
|
5
|
+
plausible. And it's still wrong, in a way that a normal test suite and a normal code review will not catch.
|
|
6
|
+
|
|
7
|
+
These are three real incidents from one live quant-research project. All three shipped — became "confirmed"
|
|
8
|
+
results the project acted on — before something caught them.
|
|
9
|
+
|
|
10
|
+
## 1. The single-anchor illusion
|
|
11
|
+
|
|
12
|
+
A covered-call strategy rebalanced every 30 trading days. Its backtest picked rebalance dates the obvious
|
|
13
|
+
way: every 30th day, starting from the first day of data. Nobody thought twice about that "first day of
|
|
14
|
+
data" choice — it's just where the data starts.
|
|
15
|
+
|
|
16
|
+
Except it isn't a neutral choice. It's a phase. Shift the starting point by a single trading day and you get
|
|
17
|
+
a genuinely different, non-overlapping sequence of ~120 entry and exit dates across a decade of data — not
|
|
18
|
+
noise around a stable number, a different simulation.
|
|
19
|
+
|
|
20
|
+
Someone eventually swept all 30 possible phases of that 30-day cycle. Sharpe ranged from 0.35 to 1.46. And
|
|
21
|
+
the one phase every prior test had used — the "obvious" one, day zero — was the single best of all 30.
|
|
22
|
+
|
|
23
|
+
Every parameter decision the project had made up to that point had been validated against the luckiest
|
|
24
|
+
number available, not a representative one. Nobody had cherry-picked it on purpose. The backtest had
|
|
25
|
+
cherry-picked it for them, silently, the moment someone wrote `data[::30]` instead of thinking about what
|
|
26
|
+
that slice actually meant.
|
|
27
|
+
|
|
28
|
+
**The fix is almost insultingly simple: run every phase, not one.** The hard part is remembering to, every
|
|
29
|
+
single time, on every candidate, forever — which is the actual reason to put it in a library instead of a
|
|
30
|
+
habit.
|
|
31
|
+
|
|
32
|
+
## 2. The bug that only breaks in one direction
|
|
33
|
+
|
|
34
|
+
Drawdown is naturally a negative number. A 20% drawdown is `-0.20`; a worse, 30% drawdown is `-0.30`. That's
|
|
35
|
+
correct and everyone knows it. It's also exactly the setup for a comparison bug that only breaks in the
|
|
36
|
+
direction that makes you happy.
|
|
37
|
+
|
|
38
|
+
Write the comparison the way it reads in English — "is the candidate's drawdown at least as good as the
|
|
39
|
+
baseline's?" — and you get `candidate_dd <= baseline_dd`. Plug in the numbers: `-0.30 <= -0.20` is `True`.
|
|
40
|
+
The candidate with the DEEPER, WORSE drawdown just passed the check.
|
|
41
|
+
|
|
42
|
+
This shipped. Three separate candidates were reported as "confirmed improvements — every metric passed" on
|
|
43
|
+
this exact line of code, before someone doing an unrelated manual double-check noticed the numbers didn't
|
|
44
|
+
smell right and traced it back.
|
|
45
|
+
|
|
46
|
+
The bug is one character's worth of intent, encoded in a symbol (`<=`) that means something different than
|
|
47
|
+
it looks like it means once you remember which way the number is stored. Nobody writing that line was
|
|
48
|
+
careless. The line just doesn't say what it looks like it says.
|
|
49
|
+
|
|
50
|
+
**The fix, again, is simple — get the direction right once, in one function everyone calls, and never let
|
|
51
|
+
anyone write the comparison inline again.** The value isn't the fix. It's that the fix has to exist
|
|
52
|
+
somewhere other than "be careful," because "be careful" already failed three times.
|
|
53
|
+
|
|
54
|
+
## 3. The result that was better on a strategy with no strategy in it
|
|
55
|
+
|
|
56
|
+
This is the one that should worry you most, because it doesn't look like a bug. It looks like a genuine,
|
|
57
|
+
exciting result.
|
|
58
|
+
|
|
59
|
+
The idea: split one strategy into several "tranches," each running the same logic but offset by a few days,
|
|
60
|
+
then blend them — average tranche 1's cycle-7 return with tranche 2's cycle-7 return, and so on — the way
|
|
61
|
+
you'd diversify a portfolio across managers who trade slightly out of phase with each other.
|
|
62
|
+
|
|
63
|
+
It looked fantastic. Sharpe up 27%. Drawdown several points shallower. A real, structural improvement, the
|
|
64
|
+
kind you build a new version of your strategy around.
|
|
65
|
+
|
|
66
|
+
Someone ran the control before believing it: apply the *exact same blending code* to plain buy-and-hold. Buy
|
|
67
|
+
plain SPY, do nothing else, blend it the same way. Buy-and-hold cannot possibly have any real "phase skill"
|
|
68
|
+
to capture — there's no strategy in it, just a price series. It "improved" by almost the same ratio. Sharpe
|
|
69
|
+
0.89 to 1.20, on doing literally nothing, blended.
|
|
70
|
+
|
|
71
|
+
What was actually happening: averaging several time-shifted windows smooths out variance across those
|
|
72
|
+
windows. It's a low-pass filter on noise, dressed up in returns notation. It has nothing to do with skill,
|
|
73
|
+
diversification, or risk reduction — it would have "worked" on a coin flip. If that one control run hadn't
|
|
74
|
+
happened, that result would have shipped as the project's best-ever finding.
|
|
75
|
+
|
|
76
|
+
**The generalizable lesson: if a technique can only be validated by looking at the strategy, it hasn't been
|
|
77
|
+
validated.** Construct the version where you know the honest answer in advance — a control, a null model, a
|
|
78
|
+
strategy-free baseline — and run your technique against *that* first. If it "wins" there too, you've found
|
|
79
|
+
an artifact, not an edge.
|
|
80
|
+
|
|
81
|
+
---
|
|
82
|
+
|
|
83
|
+
## Why these three, and not a list of forty
|
|
84
|
+
|
|
85
|
+
They're not exotic. None involves a subtle statistics error or a rare edge case in floating-point math.
|
|
86
|
+
They're the kind of mistake that survives a working code review, survives `git blame`, survives a green test
|
|
87
|
+
suite — because in each case, the code did exactly what it was written to do. The bug was in what it was
|
|
88
|
+
written to *mean*.
|
|
89
|
+
|
|
90
|
+
That's the actual argument for packaging these as automated checks instead of a mental checklist: a mental
|
|
91
|
+
checklist degrades every time you're tired, in a hurry, or excited about a number that finally looks good.
|
|
92
|
+
A function that runs anyway doesn't.
|
|
93
|
+
|
|
94
|
+
## What we built
|
|
95
|
+
|
|
96
|
+
**[AnchorTest](https://github.com/grant02339-ship-it/anchortest)** is the smallest library that runs these three
|
|
97
|
+
checks by default: anchor-average instead of trusting one phase, compare with the sign convention fixed in
|
|
98
|
+
one place, and check any blending logic against a phase-invariant control before trusting it. Plus a small
|
|
99
|
+
mutation-testing helper, because "we have tests" is a claim that deserves its own falsification check.
|
|
100
|
+
|
|
101
|
+
It's free, it's MIT-licensed, and it's about 400 lines of code — on purpose. The value isn't the amount of
|
|
102
|
+
code. It's that these three specific mistakes, having already cost one project real time and nearly cost it
|
|
103
|
+
a shipped false result, don't get to cost you the same thing.
|
|
104
|
+
|
|
105
|
+
```bash
|
|
106
|
+
pip install -e . # from the repo, until it's on PyPI
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
If you've got your own war story — a result that looked right until you ran the control that wasn't obvious
|
|
110
|
+
to run — that's the kind of issue or PR this project wants most.
|
|
@@ -0,0 +1,152 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: anchortest
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Stop trusting a single backtest run: anchor-average across every phase of your cycle, compare candidates without the drawdown sign trap, and catch blending artifacts before you ship a strategy.
|
|
5
|
+
Project-URL: Homepage, https://github.com/grant02339-ship-it/anchortest
|
|
6
|
+
Project-URL: Repository, https://github.com/grant02339-ship-it/anchortest
|
|
7
|
+
Author-email: grant02339@gmail.com
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Keywords: backtesting,finance,overfitting,quant,trading,validation
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Intended Audience :: Financial and Insurance Industry
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Topic :: Office/Business :: Financial :: Investment
|
|
16
|
+
Requires-Python: >=3.9
|
|
17
|
+
Requires-Dist: numpy>=1.24
|
|
18
|
+
Requires-Dist: pandas>=2.0
|
|
19
|
+
Provides-Extra: dev
|
|
20
|
+
Requires-Dist: pytest>=7.0; extra == 'dev'
|
|
21
|
+
Description-Content-Type: text/markdown
|
|
22
|
+
|
|
23
|
+
# AnchorTest
|
|
24
|
+
|
|
25
|
+
**Stop trusting a single backtest run.**
|
|
26
|
+
|
|
27
|
+
Most backtests report one run's Sharpe ratio, on one start date, over one universe. That number is a sample
|
|
28
|
+
of one. AnchorTest is a small, dependency-light Python library (`numpy` + `pandas`, nothing else) that runs
|
|
29
|
+
three specific checks that have each caught a real, shipped bug in real trading-strategy research:
|
|
30
|
+
|
|
31
|
+
1. **Anchor-average**, instead of trusting one lucky start date.
|
|
32
|
+
2. **Compare candidates correctly**, with the drawdown sign-convention trap fixed once, in one place.
|
|
33
|
+
3. **Catch a blending/tranching artifact** before it becomes your headline result.
|
|
34
|
+
|
|
35
|
+
Plus a small, general-purpose **mutation-testing helper**, because a green test suite doesn't tell you it
|
|
36
|
+
would have caught the bug you didn't think to write a test for.
|
|
37
|
+
|
|
38
|
+
## Why this exists
|
|
39
|
+
|
|
40
|
+
Three real incidents from one live quant-research project, each of which shipped before it was caught:
|
|
41
|
+
|
|
42
|
+
> **1. A single-anchor backtest is a sample of one.** A strategy's backtest struck rebalance dates as a
|
|
43
|
+
> fixed stride starting from wherever the data happened to begin. Sweeping every possible phase of a 30-day
|
|
44
|
+
> cycle found Sharpe ranging from 0.35 to 1.46 -- and the ONE phase every earlier test had used happened to
|
|
45
|
+
> land on literally the best of all 30. Every parameter decision made before this was caught had been
|
|
46
|
+
> validated against the luckiest possible number, not a representative one.
|
|
47
|
+
|
|
48
|
+
> **2. A sign-convention bug that shipped three false wins.** Drawdown is naturally stored as a negative
|
|
49
|
+
> number (a shallower, better drawdown is closer to zero). A comparison written the obvious way --
|
|
50
|
+
> `candidate_dd <= baseline_dd` -- silently treats a DEEPER, WORSE drawdown as a pass, because `-0.30 <=
|
|
51
|
+
> -0.20` is true. Three candidates were reported as "confirmed improvements" on this exact bug before a
|
|
52
|
+
> manual re-check caught it.
|
|
53
|
+
|
|
54
|
+
> **3. A blending result that was pure measurement artifact.** Averaging several phase-shifted copies of a
|
|
55
|
+
> strategy (offset "tranches") by taking the mean of cycle-i's return across all of them looked like a
|
|
56
|
+
> genuine diversification win: Sharpe up 27%, drawdown several points shallower. The check that caught it:
|
|
57
|
+
> running the IDENTICAL blend on plain buy-and-hold, which cannot have any real phase-dependent skill by
|
|
58
|
+
> construction. Buy-and-hold "improved" by almost the same ratio (Sharpe 0.89 -> 1.20). The blend was
|
|
59
|
+
> smoothing variance across offset windows, not reducing real risk -- and it would have shipped as the
|
|
60
|
+
> project's best result if that one control hadn't been run.
|
|
61
|
+
|
|
62
|
+
None of these are exotic mistakes. They're the kind of thing that survives code review, survives a
|
|
63
|
+
reasonable test suite, and looks exactly like a genuine result until someone runs the specific control that
|
|
64
|
+
exposes it. AnchorTest packages those controls so you run them by default, not by luck.
|
|
65
|
+
|
|
66
|
+
## Install
|
|
67
|
+
|
|
68
|
+
```bash
|
|
69
|
+
pip install anchortest # once published -- see below
|
|
70
|
+
# or, for now:
|
|
71
|
+
pip install -e .
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
Requires Python 3.9+, `numpy`, `pandas`. No `scipy`, no broker SDK, no options-pricing library -- this works
|
|
75
|
+
on any equity-curve-producing backtest, in any asset class, options or not.
|
|
76
|
+
|
|
77
|
+
## Paid audits
|
|
78
|
+
|
|
79
|
+
I'll run these checks against YOUR backtest and send back a written report: what passed, what didn't, and
|
|
80
|
+
why. **Quick Check ($199, 3 business days):** anchor-average your existing backtest, run it against the
|
|
81
|
+
strict bar, check any blending logic for the artifact pattern above. **Full Audit ($749, 7 business days):**
|
|
82
|
+
Quick Check plus a held-out/out-of-sample test designed for your specific strategy, plus a 30-minute call.
|
|
83
|
+
This is a review of your validation process, not investment advice and not a claim that any strategy is
|
|
84
|
+
profitable. Email grant02339@gmail.com to book, or see the landing page (`docs/index.html`).
|
|
85
|
+
|
|
86
|
+
## Quickstart
|
|
87
|
+
|
|
88
|
+
```python
|
|
89
|
+
from anchortest import anchor_average, compare, clears_bar
|
|
90
|
+
|
|
91
|
+
def backtest(params, start_shift=0):
|
|
92
|
+
"""Your own backtest. MUST accept start_shift and drop that many rows of your underlying
|
|
93
|
+
price data before computing anything -- see examples/random_walk_example.py for a full one."""
|
|
94
|
+
... # returns a pandas Series equity curve, indexed by date, starting at 1.0
|
|
95
|
+
|
|
96
|
+
baseline, _ = anchor_average(backtest, baseline_params, cycle_days=25, n_anchors=25)
|
|
97
|
+
candidate, _ = anchor_average(backtest, candidate_params, cycle_days=25, n_anchors=25)
|
|
98
|
+
|
|
99
|
+
delta = compare(baseline, candidate)
|
|
100
|
+
if clears_bar(delta):
|
|
101
|
+
print("candidate improves EVERY tracked metric -- worth a closer look")
|
|
102
|
+
else:
|
|
103
|
+
print("not a promotion:", delta)
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
Run `python examples/random_walk_example.py` for a runnable end-to-end demo, including a case where the
|
|
107
|
+
single-anchor (`shift=0`) result actively disagrees with the honest average.
|
|
108
|
+
|
|
109
|
+
## What's in the box
|
|
110
|
+
|
|
111
|
+
| function | catches |
|
|
112
|
+
|---|---|
|
|
113
|
+
| `anchor_average(backtest_fn, params, cycle_days, n_anchors=25)` | trusting one lucky start date |
|
|
114
|
+
| `shifts_for(cycle_days, n_anchors)` | the phase-shift arithmetic itself (deduplicates correctly when `n_anchors > cycle_days`) |
|
|
115
|
+
| `compare(baseline, candidate)` / `clears_bar(delta)` | the drawdown sign-convention trap; "5 of 6 metrics improved" being reported as a win |
|
|
116
|
+
| `check_blend_artifact(blend_fn)` / `assert_no_blend_artifact(result)` | a tranching/blending function that inflates Sharpe with no real skill |
|
|
117
|
+
| `run_mutation_suite(mutations, test_command)` / `assert_all_caught(results)` | a test suite that would not actually have caught the bug you just fixed |
|
|
118
|
+
|
|
119
|
+
Each function's docstring explains the real failure mode it generalizes from -- read them; the "why" is not
|
|
120
|
+
padding.
|
|
121
|
+
|
|
122
|
+
## Philosophy
|
|
123
|
+
|
|
124
|
+
- **Compare against a control before you trust a transform.** If you can construct a version of your data
|
|
125
|
+
where the "true" answer is known (a phase-invariant control, a random baseline, a null model), run your
|
|
126
|
+
method against it before running it on the real thing. `check_blend_artifact` is one instance of this
|
|
127
|
+
principle; it generalizes further than this library currently automates.
|
|
128
|
+
- **The strict bar exists because partial improvement has been about a coin flip.** A candidate that
|
|
129
|
+
improves mean Sharpe while its minimum-case Sharpe or recent-half Sharpe gets worse is not free money --
|
|
130
|
+
in real testing, that exact pattern has predicted a real-world regression as often as an improvement.
|
|
131
|
+
`clears_bar`'s default requires every tracked metric to improve, deliberately.
|
|
132
|
+
- **A green test suite is a claim, not a fact, until you've tried to falsify it.** `run_mutation_suite` is
|
|
133
|
+
the smallest possible version of "did this test actually assert anything, or did it just not crash."
|
|
134
|
+
|
|
135
|
+
## What this is not
|
|
136
|
+
|
|
137
|
+
- Not a backtesting engine. Bring your own `backtest_fn`; AnchorTest only tells you how much to trust its
|
|
138
|
+
output.
|
|
139
|
+
- Not investment advice, and not a signal that any particular strategy works. It's a set of falsification
|
|
140
|
+
checks -- what survives them still needs real out-of-sample and (if it trades real markets) real
|
|
141
|
+
historical-data testing before anyone should trust it with money.
|
|
142
|
+
- Not a substitute for testing out-of-sample, on assets or time periods you didn't tune on. That discipline
|
|
143
|
+
matters at least as much as anything in this library automates; there's no shortcut for it here yet.
|
|
144
|
+
|
|
145
|
+
## Contributing
|
|
146
|
+
|
|
147
|
+
Issues and PRs welcome, especially: additional artifact-detection controls, a walk-forward /
|
|
148
|
+
purged-cross-validation helper, and real-world "this caught a bug" case studies to add to the list above.
|
|
149
|
+
|
|
150
|
+
## License
|
|
151
|
+
|
|
152
|
+
MIT. See `LICENSE`.
|
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
# AnchorTest
|
|
2
|
+
|
|
3
|
+
**Stop trusting a single backtest run.**
|
|
4
|
+
|
|
5
|
+
Most backtests report one run's Sharpe ratio, on one start date, over one universe. That number is a sample
|
|
6
|
+
of one. AnchorTest is a small, dependency-light Python library (`numpy` + `pandas`, nothing else) that runs
|
|
7
|
+
three specific checks that have each caught a real, shipped bug in real trading-strategy research:
|
|
8
|
+
|
|
9
|
+
1. **Anchor-average**, instead of trusting one lucky start date.
|
|
10
|
+
2. **Compare candidates correctly**, with the drawdown sign-convention trap fixed once, in one place.
|
|
11
|
+
3. **Catch a blending/tranching artifact** before it becomes your headline result.
|
|
12
|
+
|
|
13
|
+
Plus a small, general-purpose **mutation-testing helper**, because a green test suite doesn't tell you it
|
|
14
|
+
would have caught the bug you didn't think to write a test for.
|
|
15
|
+
|
|
16
|
+
## Why this exists
|
|
17
|
+
|
|
18
|
+
Three real incidents from one live quant-research project, each of which shipped before it was caught:
|
|
19
|
+
|
|
20
|
+
> **1. A single-anchor backtest is a sample of one.** A strategy's backtest struck rebalance dates as a
|
|
21
|
+
> fixed stride starting from wherever the data happened to begin. Sweeping every possible phase of a 30-day
|
|
22
|
+
> cycle found Sharpe ranging from 0.35 to 1.46 -- and the ONE phase every earlier test had used happened to
|
|
23
|
+
> land on literally the best of all 30. Every parameter decision made before this was caught had been
|
|
24
|
+
> validated against the luckiest possible number, not a representative one.
|
|
25
|
+
|
|
26
|
+
> **2. A sign-convention bug that shipped three false wins.** Drawdown is naturally stored as a negative
|
|
27
|
+
> number (a shallower, better drawdown is closer to zero). A comparison written the obvious way --
|
|
28
|
+
> `candidate_dd <= baseline_dd` -- silently treats a DEEPER, WORSE drawdown as a pass, because `-0.30 <=
|
|
29
|
+
> -0.20` is true. Three candidates were reported as "confirmed improvements" on this exact bug before a
|
|
30
|
+
> manual re-check caught it.
|
|
31
|
+
|
|
32
|
+
> **3. A blending result that was pure measurement artifact.** Averaging several phase-shifted copies of a
|
|
33
|
+
> strategy (offset "tranches") by taking the mean of cycle-i's return across all of them looked like a
|
|
34
|
+
> genuine diversification win: Sharpe up 27%, drawdown several points shallower. The check that caught it:
|
|
35
|
+
> running the IDENTICAL blend on plain buy-and-hold, which cannot have any real phase-dependent skill by
|
|
36
|
+
> construction. Buy-and-hold "improved" by almost the same ratio (Sharpe 0.89 -> 1.20). The blend was
|
|
37
|
+
> smoothing variance across offset windows, not reducing real risk -- and it would have shipped as the
|
|
38
|
+
> project's best result if that one control hadn't been run.
|
|
39
|
+
|
|
40
|
+
None of these are exotic mistakes. They're the kind of thing that survives code review, survives a
|
|
41
|
+
reasonable test suite, and looks exactly like a genuine result until someone runs the specific control that
|
|
42
|
+
exposes it. AnchorTest packages those controls so you run them by default, not by luck.
|
|
43
|
+
|
|
44
|
+
## Install
|
|
45
|
+
|
|
46
|
+
```bash
|
|
47
|
+
pip install anchortest # once published -- see below
|
|
48
|
+
# or, for now:
|
|
49
|
+
pip install -e .
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
Requires Python 3.9+, `numpy`, `pandas`. No `scipy`, no broker SDK, no options-pricing library -- this works
|
|
53
|
+
on any equity-curve-producing backtest, in any asset class, options or not.
|
|
54
|
+
|
|
55
|
+
## Paid audits
|
|
56
|
+
|
|
57
|
+
I'll run these checks against YOUR backtest and send back a written report: what passed, what didn't, and
|
|
58
|
+
why. **Quick Check ($199, 3 business days):** anchor-average your existing backtest, run it against the
|
|
59
|
+
strict bar, check any blending logic for the artifact pattern above. **Full Audit ($749, 7 business days):**
|
|
60
|
+
Quick Check plus a held-out/out-of-sample test designed for your specific strategy, plus a 30-minute call.
|
|
61
|
+
This is a review of your validation process, not investment advice and not a claim that any strategy is
|
|
62
|
+
profitable. Email grant02339@gmail.com to book, or see the landing page (`docs/index.html`).
|
|
63
|
+
|
|
64
|
+
## Quickstart
|
|
65
|
+
|
|
66
|
+
```python
|
|
67
|
+
from anchortest import anchor_average, compare, clears_bar
|
|
68
|
+
|
|
69
|
+
def backtest(params, start_shift=0):
|
|
70
|
+
"""Your own backtest. MUST accept start_shift and drop that many rows of your underlying
|
|
71
|
+
price data before computing anything -- see examples/random_walk_example.py for a full one."""
|
|
72
|
+
... # returns a pandas Series equity curve, indexed by date, starting at 1.0
|
|
73
|
+
|
|
74
|
+
baseline, _ = anchor_average(backtest, baseline_params, cycle_days=25, n_anchors=25)
|
|
75
|
+
candidate, _ = anchor_average(backtest, candidate_params, cycle_days=25, n_anchors=25)
|
|
76
|
+
|
|
77
|
+
delta = compare(baseline, candidate)
|
|
78
|
+
if clears_bar(delta):
|
|
79
|
+
print("candidate improves EVERY tracked metric -- worth a closer look")
|
|
80
|
+
else:
|
|
81
|
+
print("not a promotion:", delta)
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
Run `python examples/random_walk_example.py` for a runnable end-to-end demo, including a case where the
|
|
85
|
+
single-anchor (`shift=0`) result actively disagrees with the honest average.
|
|
86
|
+
|
|
87
|
+
## What's in the box
|
|
88
|
+
|
|
89
|
+
| function | catches |
|
|
90
|
+
|---|---|
|
|
91
|
+
| `anchor_average(backtest_fn, params, cycle_days, n_anchors=25)` | trusting one lucky start date |
|
|
92
|
+
| `shifts_for(cycle_days, n_anchors)` | the phase-shift arithmetic itself (deduplicates correctly when `n_anchors > cycle_days`) |
|
|
93
|
+
| `compare(baseline, candidate)` / `clears_bar(delta)` | the drawdown sign-convention trap; "5 of 6 metrics improved" being reported as a win |
|
|
94
|
+
| `check_blend_artifact(blend_fn)` / `assert_no_blend_artifact(result)` | a tranching/blending function that inflates Sharpe with no real skill |
|
|
95
|
+
| `run_mutation_suite(mutations, test_command)` / `assert_all_caught(results)` | a test suite that would not actually have caught the bug you just fixed |
|
|
96
|
+
|
|
97
|
+
Each function's docstring explains the real failure mode it generalizes from -- read them; the "why" is not
|
|
98
|
+
padding.
|
|
99
|
+
|
|
100
|
+
## Philosophy
|
|
101
|
+
|
|
102
|
+
- **Compare against a control before you trust a transform.** If you can construct a version of your data
|
|
103
|
+
where the "true" answer is known (a phase-invariant control, a random baseline, a null model), run your
|
|
104
|
+
method against it before running it on the real thing. `check_blend_artifact` is one instance of this
|
|
105
|
+
principle; it generalizes further than this library currently automates.
|
|
106
|
+
- **The strict bar exists because partial improvement has been about a coin flip.** A candidate that
|
|
107
|
+
improves mean Sharpe while its minimum-case Sharpe or recent-half Sharpe gets worse is not free money --
|
|
108
|
+
in real testing, that exact pattern has predicted a real-world regression as often as an improvement.
|
|
109
|
+
`clears_bar`'s default requires every tracked metric to improve, deliberately.
|
|
110
|
+
- **A green test suite is a claim, not a fact, until you've tried to falsify it.** `run_mutation_suite` is
|
|
111
|
+
the smallest possible version of "did this test actually assert anything, or did it just not crash."
|
|
112
|
+
|
|
113
|
+
## What this is not
|
|
114
|
+
|
|
115
|
+
- Not a backtesting engine. Bring your own `backtest_fn`; AnchorTest only tells you how much to trust its
|
|
116
|
+
output.
|
|
117
|
+
- Not investment advice, and not a signal that any particular strategy works. It's a set of falsification
|
|
118
|
+
checks -- what survives them still needs real out-of-sample and (if it trades real markets) real
|
|
119
|
+
historical-data testing before anyone should trust it with money.
|
|
120
|
+
- Not a substitute for testing out-of-sample, on assets or time periods you didn't tune on. That discipline
|
|
121
|
+
matters at least as much as anything in this library automates; there's no shortcut for it here yet.
|
|
122
|
+
|
|
123
|
+
## Contributing
|
|
124
|
+
|
|
125
|
+
Issues and PRs welcome, especially: additional artifact-detection controls, a walk-forward /
|
|
126
|
+
purged-cross-validation helper, and real-world "this caught a bug" case studies to add to the list above.
|
|
127
|
+
|
|
128
|
+
## License
|
|
129
|
+
|
|
130
|
+
MIT. See `LICENSE`.
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
# Releasing
|
|
2
|
+
|
|
3
|
+
Publishing uses PyPI's **Trusted Publishing** (OIDC): GitHub proves its identity to PyPI at publish time,
|
|
4
|
+
so no API token is ever stored in this repo, in a GitHub secret, or anywhere else. `.github/workflows/publish.yml`
|
|
5
|
+
already does the automated half; the parts below need a human with a PyPI account.
|
|
6
|
+
|
|
7
|
+
## One-time setup (do this once, before the first release)
|
|
8
|
+
|
|
9
|
+
1. Create a PyPI account at [pypi.org/account/register](https://pypi.org/account/register/), if you don't
|
|
10
|
+
have one. Enable 2FA -- PyPI requires it for publishing.
|
|
11
|
+
2. Go to [pypi.org/manage/account/publishing](https://pypi.org/manage/account/publishing/) and add a
|
|
12
|
+
**pending publisher** (this works even though the `anchortest` project doesn't exist on PyPI yet):
|
|
13
|
+
- PyPI project name: `anchortest`
|
|
14
|
+
- Owner: `grant02339-ship-it`
|
|
15
|
+
- Repository name: `anchortest`
|
|
16
|
+
- Workflow name: `publish.yml`
|
|
17
|
+
- Environment name: `pypi`
|
|
18
|
+
3. That's it -- no token to copy anywhere. The first successful run of the `publish` job creates the
|
|
19
|
+
project on PyPI automatically.
|
|
20
|
+
|
|
21
|
+
## Cutting a release (every time after that)
|
|
22
|
+
|
|
23
|
+
```bash
|
|
24
|
+
cd ~/anchortest
|
|
25
|
+
# 1. bump the version
|
|
26
|
+
sed -i '' 's/version = "0.1.0"/version = "0.1.1"/' pyproject.toml # pick the real next version
|
|
27
|
+
git add pyproject.toml
|
|
28
|
+
git commit -m "Bump version to 0.1.1"
|
|
29
|
+
git push
|
|
30
|
+
|
|
31
|
+
# 2. tag and create a GitHub Release -- this is what triggers publish.yml
|
|
32
|
+
git tag v0.1.1
|
|
33
|
+
git push origin v0.1.1
|
|
34
|
+
gh release create v0.1.1 --title "v0.1.1" --generate-notes
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
`gh release create` triggers the `release: published` event, which runs the full pipeline: tests +
|
|
38
|
+
mutation check, build, then publish to PyPI. Watch it with `gh run watch` or at
|
|
39
|
+
`https://github.com/grant02339-ship-it/anchortest/actions`.
|
|
40
|
+
|
|
41
|
+
## Sanity check after the first publish
|
|
42
|
+
|
|
43
|
+
```bash
|
|
44
|
+
pip install anchortest
|
|
45
|
+
python -c "import anchortest; print(anchortest.__version__)"
|
|
46
|
+
```
|