Merge pull request #33 from ParallelLLC/dev
Browse filesMerge Claude updates from dev: algotrader 2.0 and Yahoo ingest fixes
- .github/workflows/sync-hf-space.yml +52 -0
- README.md +144 -76
- SPACE_README.md +114 -0
- agentic_ai_system/yahoo_data_stream.py +122 -12
- algotrader/__init__.py +63 -0
- algotrader/attribution.py +148 -0
- algotrader/charts.py +393 -0
- algotrader/cli.py +290 -0
- algotrader/cross_sectional.py +290 -0
- algotrader/data.py +252 -0
- algotrader/engine.py +140 -0
- algotrader/indicators.py +102 -0
- algotrader/lab.py +328 -0
- algotrader/metrics.py +157 -0
- algotrader/panel.py +283 -0
- algotrader/portfolio.py +257 -0
- algotrader/portfolio_lab.py +319 -0
- algotrader/strategies.py +345 -0
- algotrader/types.py +100 -0
- algotrader/validation/__init__.py +15 -0
- algotrader/validation/cross_permutation.py +146 -0
- algotrader/validation/deflated_sharpe.py +145 -0
- algotrader/validation/pbo.py +120 -0
- algotrader/validation/permutation.py +187 -0
- algotrader/validation/walkforward.py +233 -0
- algotrader/verdict.py +211 -0
- app.py +849 -0
- config.yaml +10 -4
- docs/AGENTIC_SYSTEM_V1.md +130 -0
- requirements-space.txt +12 -0
- scripts/deploy_hf_space.sh +65 -0
- tests/test_data_ingestion.py +2 -2
- tests/test_synthetic_data_generator.py +2 -2
- tests/test_v2_cli.py +89 -0
- tests/test_v2_engine.py +158 -0
- tests/test_v2_portfolio.py +377 -0
- tests/test_v2_strategies.py +234 -0
- tests/test_v2_validation.py +179 -0
- tests/test_yahoo_data_stream.py +169 -0
- ui/dash_app.py +2 -1
- ui/jupyter_widgets.py +3 -3
.github/workflows/sync-hf-space.yml
ADDED
|
@@ -0,0 +1,52 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
name: Sync Hugging Face Space
|
| 2 |
+
|
| 3 |
+
# Publishes app.py + the algotrader package to a Hugging Face Space.
|
| 4 |
+
#
|
| 5 |
+
# Setup (one time):
|
| 6 |
+
# 1. Create the Space at https://huggingface.co/new-space with SDK "Gradio".
|
| 7 |
+
# 2. Add repository secret HF_TOKEN — a write token from
|
| 8 |
+
# https://huggingface.co/settings/tokens
|
| 9 |
+
# 3. Add repository variable HF_SPACE_ID, e.g. "your-username/backtest-reality-check".
|
| 10 |
+
#
|
| 11 |
+
# Without those set the job is skipped, so forks and PRs are unaffected.
|
| 12 |
+
|
| 13 |
+
on:
|
| 14 |
+
push:
|
| 15 |
+
branches: [main]
|
| 16 |
+
paths:
|
| 17 |
+
- 'app.py'
|
| 18 |
+
- 'algotrader/**'
|
| 19 |
+
- 'SPACE_README.md'
|
| 20 |
+
- 'requirements-space.txt'
|
| 21 |
+
- 'tests/test_v2_*.py'
|
| 22 |
+
- '.github/workflows/sync-hf-space.yml'
|
| 23 |
+
workflow_dispatch:
|
| 24 |
+
|
| 25 |
+
jobs:
|
| 26 |
+
sync:
|
| 27 |
+
runs-on: ubuntu-latest
|
| 28 |
+
if: vars.HF_SPACE_ID != ''
|
| 29 |
+
steps:
|
| 30 |
+
- uses: actions/checkout@v4
|
| 31 |
+
|
| 32 |
+
- uses: actions/setup-python@v5
|
| 33 |
+
with:
|
| 34 |
+
python-version: '3.11'
|
| 35 |
+
|
| 36 |
+
- name: Verify the Space actually runs before publishing it
|
| 37 |
+
run: |
|
| 38 |
+
pip install --quiet -r requirements-space.txt pytest
|
| 39 |
+
python -m pytest tests/test_v2_engine.py tests/test_v2_validation.py \
|
| 40 |
+
tests/test_v2_strategies.py tests/test_v2_portfolio.py -q
|
| 41 |
+
ALGOTRADER_OFFLINE=1 python -c "import app; app.build_app(); print('Space builds OK')"
|
| 42 |
+
|
| 43 |
+
- name: Push to the Space
|
| 44 |
+
env:
|
| 45 |
+
HF_TOKEN: ${{ secrets.HF_TOKEN }}
|
| 46 |
+
run: |
|
| 47 |
+
if [ -z "$HF_TOKEN" ]; then
|
| 48 |
+
echo "HF_TOKEN secret is not set; skipping publish." >&2
|
| 49 |
+
exit 0
|
| 50 |
+
fi
|
| 51 |
+
chmod +x scripts/deploy_hf_space.sh
|
| 52 |
+
./scripts/deploy_hf_space.sh "${{ vars.HF_SPACE_ID }}"
|
README.md
CHANGED
|
@@ -1,127 +1,195 @@
|
|
| 1 |
# Algorithmic Trading
|
| 2 |
|
| 3 |
-
|
| 4 |
|
| 5 |
-
|
|
|
|
|
|
|
|
|
|
| 6 |
|
| 7 |
---
|
| 8 |
|
| 9 |
## 1. Title and Summary
|
| 10 |
|
| 11 |
**Algorithmic Trading**
|
| 12 |
-
|
| 13 |
|
| 14 |
-
GitHub `main`
|
| 15 |
|
| 16 |
**Design themes**
|
| 17 |
|
| 18 |
-
*
|
| 19 |
-
*
|
| 20 |
-
*
|
| 21 |
-
*
|
| 22 |
-
*
|
| 23 |
-
*
|
| 24 |
|
| 25 |
---
|
| 26 |
|
| 27 |
-
## 2.
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 28 |
|
| 29 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 30 |
|
| 31 |
-
|
| 32 |
-
|
| 33 |
-
|
| 34 |
-
|
| 35 |
-
|
| 36 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 37 |
|
| 38 |
-
`
|
| 39 |
|
| 40 |
-
|
| 41 |
|
| 42 |
-
* `
|
| 43 |
-
* `FinRLAgent`: PPO / A2C / DDPG / TD3 via Stable-Baselines3; persist under `models/`
|
| 44 |
-
* `ExecutionAgent` / `AlpacaBroker`: paper simulation or Alpaca market/limit orders
|
| 45 |
|
| 46 |
-
|
|
|
|
|
|
|
| 47 |
|
| 48 |
---
|
| 49 |
|
| 50 |
-
##
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 51 |
|
| 52 |
| Layer | Tools |
|
| 53 |
| ----- | ----- |
|
| 54 |
-
| Language | Python 3.11 (CI)
|
|
|
|
| 55 |
| RL | FinRL / Stable-Baselines3, Gym/Gymnasium, PyTorch |
|
| 56 |
-
|
|
| 57 |
-
| Market data | Alpaca REST; yfinance ≥ 1.0 (Yahoo) |
|
| 58 |
| Tabular | pandas, NumPy, scikit-learn |
|
| 59 |
-
| UI | Streamlit, Dash, Jupyter
|
| 60 |
-
| Deploy | Docker Compose, GitHub Actions |
|
| 61 |
-
| Tests | pytest
|
| 62 |
|
| 63 |
---
|
| 64 |
|
| 65 |
-
##
|
| 66 |
|
| 67 |
```
|
| 68 |
algorithmic_trading/
|
| 69 |
-
├──
|
| 70 |
-
├──
|
|
|
|
|
|
|
| 71 |
├── tests/
|
| 72 |
-
├──
|
| 73 |
-
├──
|
| 74 |
-
├──
|
| 75 |
-
├─
|
| 76 |
-
|
| 77 |
-
├── requirements.txt
|
| 78 |
-
├── Dockerfile
|
| 79 |
-
└── docker-compose*.yml
|
| 80 |
```
|
| 81 |
|
| 82 |
-
Branch policy: **`main`** (protected) and **`dev`** only. Do not re-enable Dependabot or the Monday `dependency-updates` workflow; those created extra branches.
|
| 83 |
-
|
| 84 |
---
|
| 85 |
|
| 86 |
-
##
|
| 87 |
|
| 88 |
-
|
| 89 |
-
|
| 90 |
-
|
| 91 |
-
|
| 92 |
-
|
| 93 |
-
|
| 94 |
-
```
|
|
|
|
|
|
|
| 95 |
|
| 96 |
-
|
| 97 |
|
| 98 |
-
|
| 99 |
-
data_source:
|
| 100 |
-
type: 'yahoo'
|
| 101 |
-
trading:
|
| 102 |
-
symbol: 'AAPL'
|
| 103 |
-
timeframe: '1d'
|
| 104 |
-
```
|
| 105 |
|
| 106 |
```bash
|
| 107 |
-
python
|
| 108 |
-
python -m
|
| 109 |
-
pytest tests/ -q
|
| 110 |
```
|
| 111 |
|
| 112 |
-
UI launchers and Docker
|
| 113 |
-
|
| 114 |
-
---
|
| 115 |
-
|
| 116 |
-
## 6. Configuration (additive Yahoo keys)
|
| 117 |
-
|
| 118 |
-
| Key | Meaning |
|
| 119 |
-
| --- | ------- |
|
| 120 |
-
| `data_source.type` | `csv` \| `synthetic` \| `alpaca` \| `yahoo` |
|
| 121 |
-
| `yahoo.start_date` / `end_date` | Historical window; clamped per Yahoo interval limits |
|
| 122 |
-
| `yahoo.auto_adjust` | Passed to `yfinance` |
|
| 123 |
-
| `execution.broker_api` | `paper` \| `alpaca_paper` \| `alpaca_live` |
|
| 124 |
-
| `finrl.algorithm` | PPO, A2C, DDPG, TD3 |
|
| 125 |
|
| 126 |
---
|
| 127 |
|
|
|
|
| 1 |
# Algorithmic Trading
|
| 2 |
|
| 3 |
+
Parallel LLC. Two layers in one repository:
|
| 4 |
|
| 5 |
+
1. **algotrader 2.0** (`algotrader/`, `app.py`): a backtester that tries to prove a rule was luck (permutation, deflated Sharpe, PBO, walk-forward, cost stress).
|
| 6 |
+
2. **Agentic v1** (`agentic_ai_system/`): FinRL policies, Yahoo or Alpaca ingest, paper/live execution, Streamlit/Dash/Jupyter UIs, Docker.
|
| 7 |
+
|
| 8 |
+
Default market data is **Yahoo Finance** (`yfinance>=1.0`), not simulated prices. The simulator exists for offline tests (`--source synthetic` or `ALGOTRADER_OFFLINE=1` with `source=auto`). Live capital still needs a separate evaluation contract. This is research tooling, not investment advice.
|
| 9 |
|
| 10 |
---
|
| 11 |
|
| 12 |
## 1. Title and Summary
|
| 13 |
|
| 14 |
**Algorithmic Trading**
|
| 15 |
+
Ingest real OHLCV, test whether a timing or cross-sectional rule survives a hostile null, optionally train a FinRL policy, size orders under position and drawdown caps, route to paper or live Alpaca.
|
| 16 |
|
| 17 |
+
GitHub keeps two branches: `main` (protected) and `dev` (integration).
|
| 18 |
|
| 19 |
**Design themes**
|
| 20 |
|
| 21 |
+
* Yahoo as the default public tape (delayed, unofficial, lookback-limited)
|
| 22 |
+
* Validation before belief: permutation, DSR, PBO/CSCV, walk-forward, 3× cost stress
|
| 23 |
+
* FinRL (PPO, A2C, DDPG, TD3) unchanged on the v1 path
|
| 24 |
+
* Alpaca optional for authenticated bars and orders; keys from the environment
|
| 25 |
+
* Synthetic GBM / regime simulator only when requested
|
| 26 |
+
* Secrets never in git
|
| 27 |
|
| 28 |
---
|
| 29 |
|
| 30 |
+
## 2. Quick start
|
| 31 |
+
|
| 32 |
+
```bash
|
| 33 |
+
git clone https://github.com/ParallelLLC/algorithmic_trading.git
|
| 34 |
+
cd algorithmic_trading
|
| 35 |
+
python -m venv .venv && source .venv/bin/activate
|
| 36 |
+
pip install -r requirements-space.txt # algotrader + Gradio
|
| 37 |
+
# or: pip install -r requirements.txt # full v1 stack (FinRL, Dash, Docker CI)
|
| 38 |
+
```
|
| 39 |
+
|
| 40 |
+
```bash
|
| 41 |
+
python app.py # Gradio, localhost:7860, Yahoo by default
|
| 42 |
+
python -m algotrader.cli lab --symbol SPY --strategy sma_cross
|
| 43 |
+
python -m algotrader.cli lab --symbol NVDA --strategy rsi_reversion --permutations 500
|
| 44 |
+
python -m agentic_ai_system.main --mode backtest --start-date 2024-01-01 --end-date 2024-12-31
|
| 45 |
+
```
|
| 46 |
+
|
| 47 |
+
`config.yaml` defaults:
|
| 48 |
|
| 49 |
+
```yaml
|
| 50 |
+
data_source:
|
| 51 |
+
type: 'yahoo'
|
| 52 |
+
trading:
|
| 53 |
+
symbol: 'AAPL'
|
| 54 |
+
timeframe: '1d' # Yahoo 1m history is ~7 days; use 1d for multi-year windows
|
| 55 |
+
yahoo:
|
| 56 |
+
auto_adjust: true # raw Close turns splits into fake crashes
|
| 57 |
+
```
|
| 58 |
|
| 59 |
+
Alpaca is opt-in: `ALPACA_API_KEY` / `ALPACA_SECRET_KEY` and `data_source.type: alpaca` or `execution.broker_api: alpaca_paper`.
|
| 60 |
+
|
| 61 |
+
---
|
| 62 |
+
|
| 63 |
+
## 3. algotrader 2.0 (validation lab)
|
| 64 |
+
|
| 65 |
+
Most backtests answer "how much would this have made?" This one asks **how much of that was luck?**
|
| 66 |
+
|
| 67 |
+
### Two labs
|
| 68 |
+
|
| 69 |
+
**The Lab** validates a timing rule on one asset. **The Portfolio Lab** validates a cross-sectional book that ranks many names.
|
| 70 |
+
|
| 71 |
+
### The four ways a backtest lies
|
| 72 |
+
|
| 73 |
+
| The lie | The test | Where |
|
| 74 |
+
| --- | --- | --- |
|
| 75 |
+
| The market had no structure to find | Monte-Carlo permutation (shuffle bar order, keep gap/high/low/body/volume) | `algotrader/validation/permutation.py` |
|
| 76 |
+
| You tried 200 things and reported the best | Deflated Sharpe Ratio | `algotrader/validation/deflated_sharpe.py` |
|
| 77 |
+
| Parameters were fitted to the past | PBO (CSCV) and walk-forward | `algotrader/validation/pbo.py`, `walkforward.py` |
|
| 78 |
+
| The edge is smaller than the costs | Cost stress at 3× friction | `algotrader/lab.py` |
|
| 79 |
+
|
| 80 |
+
Reality Score (0–100, grades A–F): significance 30%, selection 25%, walk-forward 20%, overfitting 15%, robustness 10%. The scale is harsh on purpose. Buy-and-hold and a coin-flip stay in the arena as controls.
|
| 81 |
+
|
| 82 |
+
Cross-sectional books use a **within-date weight permutation** so market correlation survives; path-shuffle is the wrong null for a long-short ranker. Survivorship is measured. Style regression (market, momentum, low-vol, reversal, liquidity) with White standard errors.
|
| 83 |
+
|
| 84 |
+
Look-ahead: `position[t] = target[t - lag]` with `lag >= 1`. Turnover is measured against drifted weights, not `|target[t]-target[t-1]|`.
|
| 85 |
+
|
| 86 |
+
```python
|
| 87 |
+
from algotrader import LabConfig, run_lab
|
| 88 |
+
|
| 89 |
+
report = run_lab(LabConfig(
|
| 90 |
+
symbol="SPY",
|
| 91 |
+
start="2015-01-01",
|
| 92 |
+
strategy="sma_cross",
|
| 93 |
+
params={"fast": 20, "slow": 100},
|
| 94 |
+
source="yahoo",
|
| 95 |
+
n_permutations=500,
|
| 96 |
+
))
|
| 97 |
+
print(report.verdict["grade"], report.permutation.p_value, report.dsr["dsr"])
|
| 98 |
+
```
|
| 99 |
+
|
| 100 |
+
```bash
|
| 101 |
+
python -m algotrader.cli strategies
|
| 102 |
+
python -m algotrader.cli lab --symbol SPY --source yahoo
|
| 103 |
+
python -m algotrader.cli portfolio --symbols SPY,QQQ,AAPL,MSFT,NVDA --strategy xs_momentum
|
| 104 |
+
python -m algotrader.cli lab --source synthetic # offline tests only
|
| 105 |
+
```
|
| 106 |
|
| 107 |
+
Single-asset zoo: `buy_and_hold`, `sma_cross`, `ema_cross`, `macd_trend`, `rsi_reversion`, `bollinger_reversion`, `donchian_breakout`, `momentum`, `vol_target_momentum`, `channel_trend`, `coin_flip`.
|
| 108 |
|
| 109 |
+
Cross-sectional: `equal_weight`, `xs_momentum`, `xs_reversal`, `low_volatility`, `xs_value_proxy`, `xs_random`.
|
| 110 |
|
| 111 |
+
**Data:** `load_ohlcv(..., source="yahoo")` downloads from Yahoo and **raises** if the download is empty. `source="auto"` is the Space fallback (cache, then simulator). `ALGOTRADER_OFFLINE=1` disables the network.
|
|
|
|
|
|
|
| 112 |
|
| 113 |
+
**HF Space:** `HF_TOKEN=hf_xxx ./scripts/deploy_hf_space.sh <user>/backtest-reality-check`. Card is `SPACE_README.md`. Tests: `python -m pytest tests/test_v2_*.py -q`.
|
| 114 |
+
|
| 115 |
+
References: Bailey & López de Prado (2014) DSR; Bailey et al. (2016) PBO; Masters (2018) permutation tests for trading systems.
|
| 116 |
|
| 117 |
---
|
| 118 |
|
| 119 |
+
## 4. Concepts and methods (v1 ingest and execution)
|
| 120 |
+
|
| 121 |
+
| Source | Default? | Failure modes |
|
| 122 |
+
| ------ | -------- | ------------- |
|
| 123 |
+
| **Yahoo** | Yes (`config.yaml`, algotrader CLI, Gradio) | Unofficial API, ~15 min delay, 1m ≈ 7 days, split-adjustment required (`auto_adjust: true`) |
|
| 124 |
+
| **Alpaca** | Optional | Auth, feed, rate limits |
|
| 125 |
+
| **CSV** | Replay | Missing path or OHLCV columns |
|
| 126 |
+
| **Synthetic** | Tests / `--source synthetic` | Not tradable edge |
|
| 127 |
+
|
| 128 |
+
`agentic_ai_system.data_ingestion.load_data` dispatches on `data_source.type`. Yahoo stream: `yahoo_data_stream.py` (clamped lookback, no incomplete bars by default).
|
| 129 |
+
|
| 130 |
+
* `StrategyAgent`: SMA, RSI, Bollinger, MACD on Close (teaching rule, not an alpha claim)
|
| 131 |
+
* `FinRLAgent`: PPO / A2C / DDPG / TD3 via Stable-Baselines3
|
| 132 |
+
* `ExecutionAgent` / `AlpacaBroker`: paper simulation or Alpaca orders
|
| 133 |
+
|
| 134 |
+
v1 `run_backtest` is a single in-sample pass unless you use algotrader walk-forward. Leakage is the null hypothesis.
|
| 135 |
+
|
| 136 |
+
---
|
| 137 |
+
|
| 138 |
+
## 5. Stack
|
| 139 |
|
| 140 |
| Layer | Tools |
|
| 141 |
| ----- | ----- |
|
| 142 |
+
| Language | Python 3.11 (CI) |
|
| 143 |
+
| Validation | algotrader (permutation, DSR, PBO, walk-forward) |
|
| 144 |
| RL | FinRL / Stable-Baselines3, Gym/Gymnasium, PyTorch |
|
| 145 |
+
| Market data | yfinance ≥ 1.0 (default); alpaca-py optional |
|
|
|
|
| 146 |
| Tabular | pandas, NumPy, scikit-learn |
|
| 147 |
+
| UI | Gradio (`app.py`); Streamlit, Dash, Jupyter (v1) |
|
| 148 |
+
| Deploy | Docker Compose, GitHub Actions, Hugging Face Space |
|
| 149 |
+
| Tests | pytest |
|
| 150 |
|
| 151 |
---
|
| 152 |
|
| 153 |
+
## 6. Structure
|
| 154 |
|
| 155 |
```
|
| 156 |
algorithmic_trading/
|
| 157 |
+
├── algotrader/ # 2.0 lab, engine, validation, strategies
|
| 158 |
+
├── app.py # Gradio Reality Check
|
| 159 |
+
├── agentic_ai_system/ # v1 FinRL, Yahoo/Alpaca ingest, execution
|
| 160 |
+
├── ui/ # Streamlit, Dash, Jupyter, WebSocket
|
| 161 |
├── tests/
|
| 162 |
+
├── docs/AGENTIC_SYSTEM_V1.md # v1 notes
|
| 163 |
+
├── config.yaml # default data_source.type: yahoo
|
| 164 |
+
├── requirements-space.txt # Space / algotrader
|
| 165 |
+
├── requirements.txt # full v1 + CI
|
| 166 |
+
└── scripts/deploy_hf_space.sh
|
|
|
|
|
|
|
|
|
|
| 167 |
```
|
| 168 |
|
|
|
|
|
|
|
| 169 |
---
|
| 170 |
|
| 171 |
+
## 7. Configuration
|
| 172 |
|
| 173 |
+
| Key | Meaning |
|
| 174 |
+
| --- | ------- |
|
| 175 |
+
| `data_source.type` | `yahoo` (default) \| `csv` \| `synthetic` \| `alpaca` |
|
| 176 |
+
| `trading.timeframe` | Mapped to Yahoo intervals; use `1d` for multi-year history |
|
| 177 |
+
| `yahoo.auto_adjust` | Split/dividend adjust (keep true) |
|
| 178 |
+
| `yahoo.emit_incomplete_bars` | Default false; forming bars are not closes |
|
| 179 |
+
| `execution.broker_api` | `paper` \| `alpaca_paper` \| `alpaca_live` |
|
| 180 |
+
| `finrl.algorithm` | PPO, A2C, DDPG, TD3 |
|
| 181 |
+
| algotrader `--source` | `yahoo` (default) \| `auto` \| `cache` \| `synthetic` |
|
| 182 |
|
| 183 |
+
---
|
| 184 |
|
| 185 |
+
## 8. Tests and ops
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 186 |
|
| 187 |
```bash
|
| 188 |
+
python -m pytest tests/test_v2_*.py -q
|
| 189 |
+
python -m pytest tests/test_yahoo_data_stream.py tests/test_data_ingestion.py -q
|
|
|
|
| 190 |
```
|
| 191 |
|
| 192 |
+
UI launchers and Docker: `UI_SETUP.md`, `DOCKER_HUB_SETUP.md`. Branch policy: `main` and `dev` only. Do not re-enable Dependabot.
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 193 |
|
| 194 |
---
|
| 195 |
|
SPACE_README.md
ADDED
|
@@ -0,0 +1,114 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
title: Backtest Reality Check
|
| 3 |
+
emoji: 🎲
|
| 4 |
+
colorFrom: blue
|
| 5 |
+
colorTo: gray
|
| 6 |
+
sdk: gradio
|
| 7 |
+
sdk_version: 5.49.1
|
| 8 |
+
app_file: app.py
|
| 9 |
+
pinned: true
|
| 10 |
+
license: apache-2.0
|
| 11 |
+
short_description: Your backtest is probably lying to you. This proves it.
|
| 12 |
+
tags:
|
| 13 |
+
- finance
|
| 14 |
+
- quantitative-finance
|
| 15 |
+
- algorithmic-trading
|
| 16 |
+
- backtesting
|
| 17 |
+
- statistics
|
| 18 |
+
- time-series
|
| 19 |
+
---
|
| 20 |
+
|
| 21 |
+
# Backtest Reality Check
|
| 22 |
+
|
| 23 |
+
**Your backtest is probably lying to you.**
|
| 24 |
+
|
| 25 |
+
Pick a market and a trading rule. This Space runs the backtest — and then spends
|
| 26 |
+
the rest of its effort trying to prove the result was luck.
|
| 27 |
+
|
| 28 |
+
Most backtesting tools answer *"how much would this have made?"*. That is the easy
|
| 29 |
+
question, and the answer is almost always flattering. This one answers the question
|
| 30 |
+
you need before risking money: **how much of that was luck?**
|
| 31 |
+
|
| 32 |
+
## The four ways a backtest lies, and the test for each
|
| 33 |
+
|
| 34 |
+
| The lie | The test |
|
| 35 |
+
|---|---|
|
| 36 |
+
| The market had no structure to find | **Permutation test** — re-run your rule on hundreds of shuffled markets |
|
| 37 |
+
| You tried 200 things and reported the best | **Deflated Sharpe Ratio** — charge for every variant you tried |
|
| 38 |
+
| The parameters were fitted to the past | **PBO + walk-forward** — does the in-sample winner keep winning? |
|
| 39 |
+
| The edge is smaller than the costs | **Cost stress test** — triple the friction and see what survives |
|
| 40 |
+
|
| 41 |
+
Each contributes to a single **Reality Score** out of 100, with a grade from A to F.
|
| 42 |
+
The scale is deliberately harsh. Most strategies people post online score below 40.
|
| 43 |
+
|
| 44 |
+
## Two labs
|
| 45 |
+
|
| 46 |
+
**The Lab** validates a timing rule on one asset. **The Portfolio Lab** validates a
|
| 47 |
+
book that ranks many names — and it gets a harder null: we keep every date's gross
|
| 48 |
+
exposure, net exposure and position count exactly as they were and randomise only
|
| 49 |
+
**which name got which weight**. A book that beats that is picking names. One that
|
| 50 |
+
doesn't was being paid for style exposure you can buy in an ETF, which the factor
|
| 51 |
+
regression measures directly.
|
| 52 |
+
|
| 53 |
+
It also measures survivorship rather than assuming it away. A universe where every
|
| 54 |
+
name is still trading after ten years was chosen after the fact, and every number
|
| 55 |
+
computed on it is an upper bound.
|
| 56 |
+
|
| 57 |
+
## Try this first
|
| 58 |
+
|
| 59 |
+
Run the **Arena** tab on `SPY`. On most markets and most date ranges, plain
|
| 60 |
+
**buy & hold** tops the leaderboard, and the **coin flip** control out-ranks
|
| 61 |
+
several respectable-looking strategies. That is not a bug in the app — it is the
|
| 62 |
+
finding.
|
| 63 |
+
|
| 64 |
+
## How the permutation test works
|
| 65 |
+
|
| 66 |
+
We take the real price series and shuffle it. Each bar's gap, high, low, body and
|
| 67 |
+
volume are kept intact, but their **order** is destroyed. The result is a market
|
| 68 |
+
with the same volatility and the same fat tails, and no exploitable structure at
|
| 69 |
+
all. Then we re-run *your exact rule* on hundreds of these shuffled markets.
|
| 70 |
+
|
| 71 |
+
If your Sharpe ratio sits comfortably inside that cloud, your rule found nothing
|
| 72 |
+
a coin-flip market would not also have handed it.
|
| 73 |
+
|
| 74 |
+
## No look-ahead, by construction
|
| 75 |
+
|
| 76 |
+
A strategy emits a target exposure at each bar's close using only data up to that
|
| 77 |
+
bar. The engine holds `position[t] = target[t - lag]` with `lag >= 1`, so a signal
|
| 78 |
+
computed on Tuesday's close cannot earn Tuesday's move. That is the single line
|
| 79 |
+
where look-ahead could enter, and the test suite asserts it directly.
|
| 80 |
+
|
| 81 |
+
## Use it from Python
|
| 82 |
+
|
| 83 |
+
```python
|
| 84 |
+
from algotrader import LabConfig, run_lab
|
| 85 |
+
|
| 86 |
+
report = run_lab(LabConfig(symbol="SPY", strategy="sma_cross"))
|
| 87 |
+
print(report.verdict["grade"], report.verdict["score"])
|
| 88 |
+
print(report.permutation.p_value, report.dsr["dsr"], report.pbo["pbo"])
|
| 89 |
+
```
|
| 90 |
+
|
| 91 |
+
Or from the command line:
|
| 92 |
+
|
| 93 |
+
```bash
|
| 94 |
+
python -m algotrader.cli lab --symbol SPY --strategy donchian_breakout --permutations 500
|
| 95 |
+
python -m algotrader.cli arena --symbol BTC-USD
|
| 96 |
+
```
|
| 97 |
+
|
| 98 |
+
## Data
|
| 99 |
+
|
| 100 |
+
Live prices come from Yahoo Finance. When the network is unavailable or rate-limited,
|
| 101 |
+
the app falls back to a deterministic market simulator with regime switching, fat
|
| 102 |
+
tails and volatility clustering — and says so, clearly, on every result. The
|
| 103 |
+
statistics remain valid; they are just measured on a simulated market.
|
| 104 |
+
|
| 105 |
+
## References
|
| 106 |
+
|
| 107 |
+
- Bailey & López de Prado (2014), *The Deflated Sharpe Ratio*
|
| 108 |
+
- Bailey, Borwein, López de Prado & Zhu (2016), *The Probability of Backtest Overfitting*
|
| 109 |
+
- Masters (2018), *Permutation and Randomization Tests for Trading System Development*
|
| 110 |
+
|
| 111 |
+
---
|
| 112 |
+
|
| 113 |
+
Apache-2.0. Research tooling, not investment advice. Nothing here is a
|
| 114 |
+
recommendation to trade.
|
agentic_ai_system/yahoo_data_stream.py
CHANGED
|
@@ -1,4 +1,5 @@
|
|
| 1 |
import logging
|
|
|
|
| 2 |
import threading
|
| 3 |
import time
|
| 4 |
from typing import Any, Callable, Dict, List, Optional
|
|
@@ -41,6 +42,26 @@ _MAX_LOOKBACK = {
|
|
| 41 |
'3mo': None,
|
| 42 |
}
|
| 43 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 44 |
|
| 45 |
class YahooDataStream:
|
| 46 |
"""
|
|
@@ -62,8 +83,13 @@ class YahooDataStream:
|
|
| 62 |
self.symbols = ['AAPL']
|
| 63 |
yahoo_cfg = config.get('yahoo', {})
|
| 64 |
self.poll_interval = int(yahoo_cfg.get('poll_interval_seconds', 60))
|
| 65 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
| 66 |
self.interval = self._map_interval(config.get('trading', {}).get('timeframe', '1d'))
|
|
|
|
| 67 |
self.data_callbacks: List[Callable] = []
|
| 68 |
self.is_connected = False
|
| 69 |
self.data_buffer: Dict[str, Dict[str, Any]] = {}
|
|
@@ -80,11 +106,21 @@ class YahooDataStream:
|
|
| 80 |
'latest_bar': None,
|
| 81 |
}
|
| 82 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 83 |
logger.info(
|
| 84 |
-
"Initialized YahooDataStream symbols=%s interval=%s poll_interval=%ss"
|
|
|
|
| 85 |
self.symbols,
|
| 86 |
self.interval,
|
| 87 |
self.poll_interval,
|
|
|
|
|
|
|
| 88 |
)
|
| 89 |
|
| 90 |
@staticmethod
|
|
@@ -102,7 +138,9 @@ class YahooDataStream:
|
|
| 102 |
return
|
| 103 |
|
| 104 |
self._stop_event.clear()
|
| 105 |
-
|
|
|
|
|
|
|
| 106 |
self._poll_thread = threading.Thread(target=self._poll_loop, name='yahoo-poll', daemon=True)
|
| 107 |
self._poll_thread.start()
|
| 108 |
self.is_connected = True
|
|
@@ -137,7 +175,8 @@ class YahooDataStream:
|
|
| 137 |
start, end = self._clamp_window(start_date, end_date, self.interval)
|
| 138 |
try:
|
| 139 |
raw = self._download(symbol, start=start, end=end, interval=self.interval)
|
| 140 |
-
df = self._normalize_ohlcv(raw)
|
|
|
|
| 141 |
if df.empty:
|
| 142 |
logger.warning("No Yahoo historical data for %s between %s and %s", symbol, start, end)
|
| 143 |
else:
|
|
@@ -173,8 +212,6 @@ class YahooDataStream:
|
|
| 173 |
}
|
| 174 |
|
| 175 |
def generate_simulated_data(self, symbol: str) -> Dict[str, Any]:
|
| 176 |
-
import random
|
| 177 |
-
|
| 178 |
latest_data = self.get_latest_data(symbol)
|
| 179 |
base_price = 150.0
|
| 180 |
if latest_data.get('latest_bar'):
|
|
@@ -197,13 +234,40 @@ class YahooDataStream:
|
|
| 197 |
return simulated_bar
|
| 198 |
|
| 199 |
def _poll_loop(self) -> None:
|
| 200 |
-
|
|
|
|
| 201 |
try:
|
| 202 |
-
self._poll_once()
|
| 203 |
except Exception as e:
|
| 204 |
logger.error("Yahoo poll loop error: %s", e, exc_info=True)
|
| 205 |
-
|
| 206 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 207 |
for symbol in self.symbols:
|
| 208 |
try:
|
| 209 |
raw = self._download(symbol, period='5d', interval=self.interval)
|
|
@@ -212,14 +276,60 @@ class YahooDataStream:
|
|
| 212 |
logger.warning("Yahoo poll returned no bars for %s", symbol)
|
| 213 |
continue
|
| 214 |
self._ingest_new_bars(symbol, df)
|
|
|
|
| 215 |
except Exception as e:
|
| 216 |
logger.error("Yahoo poll failed for %s: %s", symbol, e)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 217 |
|
| 218 |
def _ingest_new_bars(self, symbol: str, df: pd.DataFrame) -> None:
|
|
|
|
| 219 |
last_ts = self._last_bar_ts.get(symbol)
|
| 220 |
-
rows = df
|
| 221 |
if last_ts is not None:
|
| 222 |
-
rows =
|
| 223 |
if rows.empty:
|
| 224 |
return
|
| 225 |
|
|
|
|
| 1 |
import logging
|
| 2 |
+
import random
|
| 3 |
import threading
|
| 4 |
import time
|
| 5 |
from typing import Any, Callable, Dict, List, Optional
|
|
|
|
| 42 |
'3mo': None,
|
| 43 |
}
|
| 44 |
|
| 45 |
+
# How long each bar covers. Used to tell a finished bar from the one still
|
| 46 |
+
# forming right now -- see _drop_incomplete.
|
| 47 |
+
_INTERVAL_DURATION = {
|
| 48 |
+
'1m': pd.Timedelta(minutes=1),
|
| 49 |
+
'2m': pd.Timedelta(minutes=2),
|
| 50 |
+
'5m': pd.Timedelta(minutes=5),
|
| 51 |
+
'15m': pd.Timedelta(minutes=15),
|
| 52 |
+
'30m': pd.Timedelta(minutes=30),
|
| 53 |
+
'60m': pd.Timedelta(hours=1),
|
| 54 |
+
'90m': pd.Timedelta(minutes=90),
|
| 55 |
+
'1h': pd.Timedelta(hours=1),
|
| 56 |
+
'1d': pd.Timedelta(days=1),
|
| 57 |
+
'5d': pd.Timedelta(days=5),
|
| 58 |
+
'1wk': pd.Timedelta(weeks=1),
|
| 59 |
+
}
|
| 60 |
+
|
| 61 |
+
# A daily-or-slower bar that moves more than this is almost always an
|
| 62 |
+
# unadjusted split rather than a real move (NVDA's 2024 10:1 shows up as -90%).
|
| 63 |
+
_SPLIT_SUSPECT_MOVE = 0.35
|
| 64 |
+
|
| 65 |
|
| 66 |
class YahooDataStream:
|
| 67 |
"""
|
|
|
|
| 83 |
self.symbols = ['AAPL']
|
| 84 |
yahoo_cfg = config.get('yahoo', {})
|
| 85 |
self.poll_interval = int(yahoo_cfg.get('poll_interval_seconds', 60))
|
| 86 |
+
# Adjusted by default. With auto_adjust off, Yahoo returns raw Close and
|
| 87 |
+
# every split reads as a crash: NVDA's June 2024 10:1 becomes a -90% bar.
|
| 88 |
+
self.auto_adjust = bool(yahoo_cfg.get('auto_adjust', True))
|
| 89 |
+
self.emit_incomplete_bars = bool(yahoo_cfg.get('emit_incomplete_bars', False))
|
| 90 |
+
self.max_backoff = int(yahoo_cfg.get('max_backoff_seconds', 900))
|
| 91 |
self.interval = self._map_interval(config.get('trading', {}).get('timeframe', '1d'))
|
| 92 |
+
self._consecutive_failures = 0
|
| 93 |
self.data_callbacks: List[Callable] = []
|
| 94 |
self.is_connected = False
|
| 95 |
self.data_buffer: Dict[str, Dict[str, Any]] = {}
|
|
|
|
| 106 |
'latest_bar': None,
|
| 107 |
}
|
| 108 |
|
| 109 |
+
if not self.auto_adjust:
|
| 110 |
+
logger.warning(
|
| 111 |
+
"yahoo.auto_adjust is false: prices are NOT split- or dividend-adjusted. "
|
| 112 |
+
"Every split will appear as a large single-bar loss and any backtest "
|
| 113 |
+
"spanning one will be wrong."
|
| 114 |
+
)
|
| 115 |
+
|
| 116 |
logger.info(
|
| 117 |
+
"Initialized YahooDataStream symbols=%s interval=%s poll_interval=%ss "
|
| 118 |
+
"auto_adjust=%s emit_incomplete_bars=%s",
|
| 119 |
self.symbols,
|
| 120 |
self.interval,
|
| 121 |
self.poll_interval,
|
| 122 |
+
self.auto_adjust,
|
| 123 |
+
self.emit_incomplete_bars,
|
| 124 |
)
|
| 125 |
|
| 126 |
@staticmethod
|
|
|
|
| 138 |
return
|
| 139 |
|
| 140 |
self._stop_event.clear()
|
| 141 |
+
# Seed the backoff from the first attempt: if we are already being
|
| 142 |
+
# throttled, the loop should start backed off rather than hammering.
|
| 143 |
+
self._consecutive_failures = 0 if self._poll_once() else 1
|
| 144 |
self._poll_thread = threading.Thread(target=self._poll_loop, name='yahoo-poll', daemon=True)
|
| 145 |
self._poll_thread.start()
|
| 146 |
self.is_connected = True
|
|
|
|
| 175 |
start, end = self._clamp_window(start_date, end_date, self.interval)
|
| 176 |
try:
|
| 177 |
raw = self._download(symbol, start=start, end=end, interval=self.interval)
|
| 178 |
+
df = self._drop_incomplete(self._normalize_ohlcv(raw))
|
| 179 |
+
self._warn_if_unadjusted(symbol, df)
|
| 180 |
if df.empty:
|
| 181 |
logger.warning("No Yahoo historical data for %s between %s and %s", symbol, start, end)
|
| 182 |
else:
|
|
|
|
| 212 |
}
|
| 213 |
|
| 214 |
def generate_simulated_data(self, symbol: str) -> Dict[str, Any]:
|
|
|
|
|
|
|
| 215 |
latest_data = self.get_latest_data(symbol)
|
| 216 |
base_price = 150.0
|
| 217 |
if latest_data.get('latest_bar'):
|
|
|
|
| 234 |
return simulated_bar
|
| 235 |
|
| 236 |
def _poll_loop(self) -> None:
|
| 237 |
+
delay = self.poll_interval
|
| 238 |
+
while not self._stop_event.wait(delay):
|
| 239 |
try:
|
| 240 |
+
succeeded = self._poll_once()
|
| 241 |
except Exception as e:
|
| 242 |
logger.error("Yahoo poll loop error: %s", e, exc_info=True)
|
| 243 |
+
succeeded = False
|
| 244 |
+
self._consecutive_failures = 0 if succeeded else self._consecutive_failures + 1
|
| 245 |
+
delay = self._next_delay()
|
| 246 |
+
|
| 247 |
+
def _next_delay(self) -> float:
|
| 248 |
+
"""Poll interval, backed off exponentially while Yahoo is refusing us.
|
| 249 |
+
|
| 250 |
+
Yahoo rate-limits aggressively and an unofficial API gives no
|
| 251 |
+
Retry-After, so a fixed interval just keeps you throttled. Jitter stops
|
| 252 |
+
several symbols (or several deployments) resynchronising after an outage.
|
| 253 |
+
"""
|
| 254 |
+
if self._consecutive_failures == 0:
|
| 255 |
+
base = float(self.poll_interval)
|
| 256 |
+
else:
|
| 257 |
+
base = min(
|
| 258 |
+
self.poll_interval * (2 ** self._consecutive_failures),
|
| 259 |
+
float(self.max_backoff),
|
| 260 |
+
)
|
| 261 |
+
logger.warning(
|
| 262 |
+
"Yahoo poll failed %s time(s) in a row; next attempt in ~%.0fs",
|
| 263 |
+
self._consecutive_failures,
|
| 264 |
+
base,
|
| 265 |
+
)
|
| 266 |
+
return max(1.0, base * random.uniform(0.8, 1.2))
|
| 267 |
+
|
| 268 |
+
def _poll_once(self) -> bool:
|
| 269 |
+
"""Fetch and ingest one round of bars. Returns True if any symbol succeeded."""
|
| 270 |
+
any_success = False
|
| 271 |
for symbol in self.symbols:
|
| 272 |
try:
|
| 273 |
raw = self._download(symbol, period='5d', interval=self.interval)
|
|
|
|
| 276 |
logger.warning("Yahoo poll returned no bars for %s", symbol)
|
| 277 |
continue
|
| 278 |
self._ingest_new_bars(symbol, df)
|
| 279 |
+
any_success = True
|
| 280 |
except Exception as e:
|
| 281 |
logger.error("Yahoo poll failed for %s: %s", symbol, e)
|
| 282 |
+
return any_success
|
| 283 |
+
|
| 284 |
+
def _warn_if_unadjusted(self, symbol: str, df: pd.DataFrame) -> int:
|
| 285 |
+
"""Flag single-bar moves that look like unadjusted corporate actions.
|
| 286 |
+
|
| 287 |
+
This is a backstop rather than the fix -- the fix is auto_adjust. But a
|
| 288 |
+
split slipping through silently corrupts every downstream number, so it
|
| 289 |
+
is worth naming the dates rather than letting a strategy trade them.
|
| 290 |
+
Returns the number of suspicious bars found.
|
| 291 |
+
"""
|
| 292 |
+
duration = _INTERVAL_DURATION.get(self.interval)
|
| 293 |
+
if df.empty or len(df) < 2 or duration is None or duration < pd.Timedelta(days=1):
|
| 294 |
+
return 0
|
| 295 |
+
moves = df['close'].pct_change()
|
| 296 |
+
suspects = df.loc[moves.abs() > _SPLIT_SUSPECT_MOVE, 'timestamp']
|
| 297 |
+
if len(suspects):
|
| 298 |
+
dates = ', '.join(str(pd.Timestamp(t).date()) for t in suspects.head(5))
|
| 299 |
+
logger.warning(
|
| 300 |
+
"%s has %s bar(s) moving more than %.0f%% (%s). On a liquid name that is "
|
| 301 |
+
"usually an unadjusted split, not a real move — check yahoo.auto_adjust.",
|
| 302 |
+
symbol,
|
| 303 |
+
len(suspects),
|
| 304 |
+
_SPLIT_SUSPECT_MOVE * 100,
|
| 305 |
+
dates,
|
| 306 |
+
)
|
| 307 |
+
return int(len(suspects))
|
| 308 |
+
|
| 309 |
+
def _drop_incomplete(self, df: pd.DataFrame) -> pd.DataFrame:
|
| 310 |
+
"""Remove the bar that is still forming.
|
| 311 |
+
|
| 312 |
+
Yahoo returns the in-progress period as an ordinary row. Emitting it
|
| 313 |
+
would hand the strategy a close that has not happened yet, and because
|
| 314 |
+
the watermark advances past it, the finished version never arrives.
|
| 315 |
+
"""
|
| 316 |
+
if self.emit_incomplete_bars or df.empty:
|
| 317 |
+
return df
|
| 318 |
+
duration = _INTERVAL_DURATION.get(self.interval)
|
| 319 |
+
if duration is None:
|
| 320 |
+
return df
|
| 321 |
+
now = pd.Timestamp.now(tz='UTC').tz_convert(None)
|
| 322 |
+
complete = df[df['timestamp'] + duration <= now]
|
| 323 |
+
dropped = len(df) - len(complete)
|
| 324 |
+
if dropped:
|
| 325 |
+
logger.debug("Dropped %s in-progress %s bar(s)", dropped, self.interval)
|
| 326 |
+
return complete
|
| 327 |
|
| 328 |
def _ingest_new_bars(self, symbol: str, df: pd.DataFrame) -> None:
|
| 329 |
+
rows = self._drop_incomplete(df)
|
| 330 |
last_ts = self._last_bar_ts.get(symbol)
|
|
|
|
| 331 |
if last_ts is not None:
|
| 332 |
+
rows = rows[rows['timestamp'] > last_ts]
|
| 333 |
if rows.empty:
|
| 334 |
return
|
| 335 |
|
algotrader/__init__.py
ADDED
|
@@ -0,0 +1,63 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""algotrader 2.0 — a backtester that tries to prove itself wrong.
|
| 2 |
+
|
| 3 |
+
Most backtesting libraries answer "how much would this have made?". This one
|
| 4 |
+
answers the question that actually matters before you risk money: "how much of
|
| 5 |
+
that was luck?"
|
| 6 |
+
|
| 7 |
+
Quick start::
|
| 8 |
+
|
| 9 |
+
from algotrader import LabConfig, run_lab
|
| 10 |
+
|
| 11 |
+
report = run_lab(LabConfig(symbol="SPY", strategy="sma_cross"))
|
| 12 |
+
print(report.verdict["verdict"])
|
| 13 |
+
"""
|
| 14 |
+
|
| 15 |
+
from .attribution import build_style_factors, factor_attribution
|
| 16 |
+
from .cross_sectional import XS_REGISTRY, get_xs_strategy, list_xs_strategies
|
| 17 |
+
from .data import load_ohlcv, simulate_ohlcv
|
| 18 |
+
from .engine import run_backtest
|
| 19 |
+
from .lab import LabConfig, LabReport, run_arena, run_lab
|
| 20 |
+
from .metrics import compute_metrics
|
| 21 |
+
from .panel import Panel, load_panel
|
| 22 |
+
from .portfolio import rebalance_schedule, run_portfolio_backtest
|
| 23 |
+
from .portfolio_lab import PortfolioLabConfig, PortfolioLabReport, run_portfolio_arena, run_portfolio_lab
|
| 24 |
+
from .strategies import REGISTRY, get_strategy, list_strategies
|
| 25 |
+
from .types import BacktestResult, CostModel, MarketData
|
| 26 |
+
from .verdict import reality_score
|
| 27 |
+
|
| 28 |
+
__version__ = "2.1.0"
|
| 29 |
+
|
| 30 |
+
__all__ = [
|
| 31 |
+
"__version__",
|
| 32 |
+
# single asset
|
| 33 |
+
"LabConfig",
|
| 34 |
+
"LabReport",
|
| 35 |
+
"run_lab",
|
| 36 |
+
"run_arena",
|
| 37 |
+
"run_backtest",
|
| 38 |
+
"get_strategy",
|
| 39 |
+
"list_strategies",
|
| 40 |
+
"REGISTRY",
|
| 41 |
+
# multi asset
|
| 42 |
+
"Panel",
|
| 43 |
+
"load_panel",
|
| 44 |
+
"run_portfolio_backtest",
|
| 45 |
+
"rebalance_schedule",
|
| 46 |
+
"PortfolioLabConfig",
|
| 47 |
+
"PortfolioLabReport",
|
| 48 |
+
"run_portfolio_lab",
|
| 49 |
+
"run_portfolio_arena",
|
| 50 |
+
"get_xs_strategy",
|
| 51 |
+
"list_xs_strategies",
|
| 52 |
+
"XS_REGISTRY",
|
| 53 |
+
"build_style_factors",
|
| 54 |
+
"factor_attribution",
|
| 55 |
+
# shared
|
| 56 |
+
"compute_metrics",
|
| 57 |
+
"load_ohlcv",
|
| 58 |
+
"simulate_ohlcv",
|
| 59 |
+
"BacktestResult",
|
| 60 |
+
"CostModel",
|
| 61 |
+
"MarketData",
|
| 62 |
+
"reality_score",
|
| 63 |
+
]
|
algotrader/attribution.py
ADDED
|
@@ -0,0 +1,148 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Style attribution: is this alpha, or is it beta you could have bought cheaply?
|
| 2 |
+
|
| 3 |
+
The most common way a strategy is oversold is not fraud or overfitting — it is
|
| 4 |
+
that the "edge" is a well-known risk premium wearing a new name. A book that
|
| 5 |
+
loads on market beta, or on momentum, or on low-volatility, will produce a
|
| 6 |
+
respectable Sharpe and an exciting story, and you can buy the same exposure in
|
| 7 |
+
an ETF for a few basis points.
|
| 8 |
+
|
| 9 |
+
So we regress the strategy's returns on style factors built from the panel
|
| 10 |
+
itself and ask what is left over. If the intercept is not distinguishable from
|
| 11 |
+
zero, the strategy has no alpha — however good its Sharpe looked.
|
| 12 |
+
|
| 13 |
+
Factors are constructed from the universe under test rather than downloaded,
|
| 14 |
+
which keeps this offline and self-consistent. The tradeoff is that they are
|
| 15 |
+
proxies: without fundamentals there is no true value or size factor, so those
|
| 16 |
+
are labelled honestly as what they are.
|
| 17 |
+
"""
|
| 18 |
+
|
| 19 |
+
from __future__ import annotations
|
| 20 |
+
|
| 21 |
+
from typing import Dict, Optional
|
| 22 |
+
|
| 23 |
+
import numpy as np
|
| 24 |
+
import pandas as pd
|
| 25 |
+
|
| 26 |
+
from .panel import Panel
|
| 27 |
+
from .portfolio import run_portfolio_backtest
|
| 28 |
+
from .types import CostModel
|
| 29 |
+
|
| 30 |
+
__all__ = ["build_style_factors", "factor_attribution"]
|
| 31 |
+
|
| 32 |
+
_FREE = CostModel(0.0, 0.0, 0.0) # factors are theoretical portfolios
|
| 33 |
+
|
| 34 |
+
|
| 35 |
+
def _long_short(panel: Panel, scores: pd.DataFrame, rebalance: int = 21) -> pd.Series:
|
| 36 |
+
from .cross_sectional import _rebalance_hold, scores_to_weights
|
| 37 |
+
|
| 38 |
+
weights = _rebalance_hold(scores_to_weights(scores), rebalance)
|
| 39 |
+
return run_portfolio_backtest(panel, weights, costs=_FREE).returns
|
| 40 |
+
|
| 41 |
+
|
| 42 |
+
def build_style_factors(panel: Panel, rebalance: int = 21) -> pd.DataFrame:
|
| 43 |
+
"""Construct market and style factor returns from the panel itself."""
|
| 44 |
+
close = panel.close
|
| 45 |
+
listed = close.notna()
|
| 46 |
+
counts = listed.sum(axis=1).replace(0, np.nan)
|
| 47 |
+
|
| 48 |
+
equal = listed.astype(float).div(counts, axis=0).fillna(0.0)
|
| 49 |
+
market = run_portfolio_backtest(panel, equal, costs=_FREE).returns
|
| 50 |
+
|
| 51 |
+
factors = {
|
| 52 |
+
"market": market,
|
| 53 |
+
"momentum": _long_short(panel, close.shift(20) / close.shift(250) - 1.0, rebalance),
|
| 54 |
+
"low_vol": _long_short(panel, -close.pct_change().rolling(60, min_periods=60).std(), rebalance),
|
| 55 |
+
"reversal": _long_short(panel, -close.pct_change(5), rebalance),
|
| 56 |
+
# Dollar volume is a liquidity proxy, not market cap. Named accordingly.
|
| 57 |
+
"liquidity": _long_short(panel, -np.log(panel.dollar_volume().replace(0, np.nan)), rebalance),
|
| 58 |
+
}
|
| 59 |
+
return pd.DataFrame(factors).reindex(panel.index).fillna(0.0)
|
| 60 |
+
|
| 61 |
+
|
| 62 |
+
def _white_standard_errors(x: np.ndarray, residuals: np.ndarray, xtx_inv: np.ndarray) -> np.ndarray:
|
| 63 |
+
"""Heteroskedasticity-robust (White) standard errors.
|
| 64 |
+
|
| 65 |
+
Return series are famously heteroskedastic — volatility clusters — and
|
| 66 |
+
classical standard errors would overstate the significance of alpha.
|
| 67 |
+
"""
|
| 68 |
+
meat = x.T @ (x * (residuals**2)[:, None])
|
| 69 |
+
covariance = xtx_inv @ meat @ xtx_inv
|
| 70 |
+
return np.sqrt(np.maximum(np.diag(covariance), 0.0))
|
| 71 |
+
|
| 72 |
+
|
| 73 |
+
def factor_attribution(
|
| 74 |
+
returns: pd.Series,
|
| 75 |
+
factors: pd.DataFrame,
|
| 76 |
+
periods_per_year: int = 252,
|
| 77 |
+
alpha_t_threshold: float = 2.0,
|
| 78 |
+
) -> Dict[str, object]:
|
| 79 |
+
"""Regress strategy returns on style factors; report annualised alpha and betas."""
|
| 80 |
+
aligned = pd.concat([returns.rename("strategy"), factors], axis=1).dropna()
|
| 81 |
+
if len(aligned) < 60 or factors.shape[1] == 0:
|
| 82 |
+
return {"available": False, "note": "Not enough overlapping observations for attribution."}
|
| 83 |
+
|
| 84 |
+
y = aligned["strategy"].to_numpy(dtype=float)
|
| 85 |
+
names = list(factors.columns)
|
| 86 |
+
x = np.column_stack([np.ones(len(aligned))] + [aligned[c].to_numpy(dtype=float) for c in names])
|
| 87 |
+
|
| 88 |
+
# Drop factors that are constant or collinear; a singular fit is worse than
|
| 89 |
+
# a smaller one.
|
| 90 |
+
keep = [0] + [i + 1 for i, c in enumerate(names) if aligned[c].std() > 1e-12]
|
| 91 |
+
x = x[:, keep]
|
| 92 |
+
names = [names[i - 1] for i in keep[1:]]
|
| 93 |
+
|
| 94 |
+
try:
|
| 95 |
+
xtx_inv = np.linalg.pinv(x.T @ x)
|
| 96 |
+
except np.linalg.LinAlgError: # pragma: no cover - pinv rarely fails
|
| 97 |
+
return {"available": False, "note": "Factor matrix is singular."}
|
| 98 |
+
|
| 99 |
+
beta = xtx_inv @ x.T @ y
|
| 100 |
+
fitted = x @ beta
|
| 101 |
+
residuals = y - fitted
|
| 102 |
+
errors = _white_standard_errors(x, residuals, xtx_inv)
|
| 103 |
+
t_stats = np.divide(beta, errors, out=np.zeros_like(beta), where=errors > 0)
|
| 104 |
+
|
| 105 |
+
ss_res = float(residuals @ residuals)
|
| 106 |
+
ss_tot = float(((y - y.mean()) ** 2).sum())
|
| 107 |
+
r_squared = 1.0 - ss_res / ss_tot if ss_tot > 0 else 0.0
|
| 108 |
+
|
| 109 |
+
alpha_period = float(beta[0])
|
| 110 |
+
alpha_annual = alpha_period * periods_per_year
|
| 111 |
+
alpha_t = float(t_stats[0])
|
| 112 |
+
significant = bool(abs(alpha_t) >= alpha_t_threshold and alpha_annual > 0)
|
| 113 |
+
|
| 114 |
+
betas = {name: float(b) for name, b in zip(names, beta[1:])}
|
| 115 |
+
beta_ts = {name: float(t) for name, t in zip(names, t_stats[1:])}
|
| 116 |
+
dominant = max(betas, key=lambda k: abs(betas[k])) if betas else None
|
| 117 |
+
|
| 118 |
+
if significant:
|
| 119 |
+
note = (
|
| 120 |
+
f"Alpha of {alpha_annual:.1%} a year survives the style regression "
|
| 121 |
+
f"(t = {alpha_t:.1f}). Something here is not explained by market, momentum, "
|
| 122 |
+
"low-volatility, reversal or liquidity exposure."
|
| 123 |
+
)
|
| 124 |
+
else:
|
| 125 |
+
explained = (
|
| 126 |
+
f" Most of the variation is {dominant} exposure (beta {betas[dominant]:.2f})."
|
| 127 |
+
if dominant
|
| 128 |
+
else ""
|
| 129 |
+
)
|
| 130 |
+
note = (
|
| 131 |
+
f"Alpha is {alpha_annual:.1%} a year with t = {alpha_t:.1f}, which is not "
|
| 132 |
+
f"distinguishable from zero. The style factors explain {r_squared:.0%} of the "
|
| 133 |
+
f"returns.{explained} You can buy that exposure far more cheaply than by "
|
| 134 |
+
"running this strategy."
|
| 135 |
+
)
|
| 136 |
+
|
| 137 |
+
return {
|
| 138 |
+
"available": True,
|
| 139 |
+
"alpha_annual": alpha_annual,
|
| 140 |
+
"alpha_t_stat": alpha_t,
|
| 141 |
+
"alpha_significant": significant,
|
| 142 |
+
"betas": betas,
|
| 143 |
+
"beta_t_stats": beta_ts,
|
| 144 |
+
"r_squared": float(r_squared),
|
| 145 |
+
"dominant_factor": dominant,
|
| 146 |
+
"n_obs": int(len(aligned)),
|
| 147 |
+
"note": note,
|
| 148 |
+
}
|
algotrader/charts.py
ADDED
|
@@ -0,0 +1,393 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Plotly figures for the Lab.
|
| 2 |
+
|
| 3 |
+
Colour system (dark surface, validated for CVD separation):
|
| 4 |
+
|
| 5 |
+
* **blue** is always *your strategy's honest result* — the realised equity
|
| 6 |
+
curve, the out-of-sample fold, the observed Sharpe.
|
| 7 |
+
* **orange** is always *the thing it is measured against* — buy & hold, the
|
| 8 |
+
in-sample fold, the null distribution.
|
| 9 |
+
|
| 10 |
+
Holding that mapping across every figure means a reader learns it once.
|
| 11 |
+
"""
|
| 12 |
+
|
| 13 |
+
from __future__ import annotations
|
| 14 |
+
|
| 15 |
+
from typing import Dict, Optional
|
| 16 |
+
|
| 17 |
+
import numpy as np
|
| 18 |
+
import pandas as pd
|
| 19 |
+
import plotly.graph_objects as go
|
| 20 |
+
|
| 21 |
+
SURFACE = "#1a1a19"
|
| 22 |
+
PAGE = "#0d0d0d"
|
| 23 |
+
INK = "#ffffff"
|
| 24 |
+
INK_SECONDARY = "#c3c2b7"
|
| 25 |
+
INK_MUTED = "#898781"
|
| 26 |
+
GRID = "#2c2c2a"
|
| 27 |
+
AXIS = "#383835"
|
| 28 |
+
|
| 29 |
+
SUBJECT = "#3987e5" # categorical slot 1
|
| 30 |
+
REFERENCE = "#d95926" # categorical slot 2
|
| 31 |
+
NEGATIVE = "#e66767" # negative arm of the diverging pair (drawdowns)
|
| 32 |
+
|
| 33 |
+
FONT = 'system-ui, -apple-system, "Segoe UI", sans-serif'
|
| 34 |
+
|
| 35 |
+
_EMPTY_NOTE = "Run an analysis to populate this chart."
|
| 36 |
+
|
| 37 |
+
|
| 38 |
+
def _base_layout(title: str, height: int = 340, **kwargs) -> dict:
|
| 39 |
+
return dict(
|
| 40 |
+
title=dict(text=title, font=dict(size=15, color=INK), x=0, xanchor="left", pad=dict(b=8)),
|
| 41 |
+
paper_bgcolor=PAGE,
|
| 42 |
+
plot_bgcolor=SURFACE,
|
| 43 |
+
font=dict(family=FONT, size=12, color=INK_SECONDARY),
|
| 44 |
+
height=height,
|
| 45 |
+
margin=dict(l=56, r=24, t=48, b=40),
|
| 46 |
+
hovermode="x unified",
|
| 47 |
+
hoverlabel=dict(bgcolor=SURFACE, bordercolor=AXIS, font=dict(color=INK, family=FONT)),
|
| 48 |
+
xaxis=dict(gridcolor=GRID, linecolor=AXIS, zeroline=False, tickfont=dict(color=INK_MUTED)),
|
| 49 |
+
yaxis=dict(gridcolor=GRID, linecolor=AXIS, zeroline=False, tickfont=dict(color=INK_MUTED)),
|
| 50 |
+
legend=dict(
|
| 51 |
+
orientation="h", yanchor="bottom", y=1.02, xanchor="left", x=0,
|
| 52 |
+
font=dict(color=INK_SECONDARY, size=11), bgcolor="rgba(0,0,0,0)",
|
| 53 |
+
),
|
| 54 |
+
**kwargs,
|
| 55 |
+
)
|
| 56 |
+
|
| 57 |
+
|
| 58 |
+
def empty_figure(message: str = _EMPTY_NOTE, height: int = 340) -> go.Figure:
|
| 59 |
+
fig = go.Figure()
|
| 60 |
+
fig.update_layout(**_base_layout("", height=height))
|
| 61 |
+
fig.update_xaxes(visible=False)
|
| 62 |
+
fig.update_yaxes(visible=False)
|
| 63 |
+
fig.add_annotation(
|
| 64 |
+
text=message, showarrow=False, xref="paper", yref="paper", x=0.5, y=0.5,
|
| 65 |
+
font=dict(color=INK_MUTED, size=13),
|
| 66 |
+
)
|
| 67 |
+
return fig
|
| 68 |
+
|
| 69 |
+
|
| 70 |
+
def equity_chart(report, benchmark_label: str = "Buy & hold") -> go.Figure:
|
| 71 |
+
"""Strategy equity against its benchmark, both indexed to the same start."""
|
| 72 |
+
bt = report.backtest
|
| 73 |
+
strat = bt.equity / bt.equity.iloc[0] * 100.0
|
| 74 |
+
bench = bt.benchmark_equity / bt.benchmark_equity.iloc[0] * 100.0
|
| 75 |
+
|
| 76 |
+
fig = go.Figure()
|
| 77 |
+
fig.add_trace(
|
| 78 |
+
go.Scatter(
|
| 79 |
+
x=bench.index, y=bench.to_numpy(), name=benchmark_label, mode="lines",
|
| 80 |
+
line=dict(color=REFERENCE, width=2, dash="dash"),
|
| 81 |
+
hovertemplate=benchmark_label + " %{y:.1f}<extra></extra>",
|
| 82 |
+
)
|
| 83 |
+
)
|
| 84 |
+
fig.add_trace(
|
| 85 |
+
go.Scatter(
|
| 86 |
+
x=strat.index, y=strat.to_numpy(), name=report.strategy.name, mode="lines",
|
| 87 |
+
line=dict(color=SUBJECT, width=2),
|
| 88 |
+
hovertemplate=report.strategy.name + " %{y:.1f}<extra></extra>",
|
| 89 |
+
)
|
| 90 |
+
)
|
| 91 |
+
|
| 92 |
+
# Direct-label the two endpoints; the axis and tooltip carry everything else.
|
| 93 |
+
for series, color, label in ((strat, SUBJECT, report.strategy.name), (bench, REFERENCE, benchmark_label)):
|
| 94 |
+
fig.add_annotation(
|
| 95 |
+
x=series.index[-1], y=float(series.iloc[-1]),
|
| 96 |
+
text=f" {label}: {series.iloc[-1]:.0f}", showarrow=False,
|
| 97 |
+
xanchor="left", font=dict(color=color, size=11),
|
| 98 |
+
)
|
| 99 |
+
|
| 100 |
+
fig.update_layout(**_base_layout("Growth of 100 (net of costs)", height=360))
|
| 101 |
+
fig.update_layout(margin=dict(l=56, r=140, t=48, b=40))
|
| 102 |
+
return fig
|
| 103 |
+
|
| 104 |
+
|
| 105 |
+
def drawdown_chart(report) -> go.Figure:
|
| 106 |
+
"""Underwater plot — how deep, and for how long."""
|
| 107 |
+
from .metrics import drawdown_series
|
| 108 |
+
|
| 109 |
+
dd = drawdown_series(report.backtest.equity) * 100.0
|
| 110 |
+
fig = go.Figure(
|
| 111 |
+
go.Scatter(
|
| 112 |
+
x=dd.index, y=dd.to_numpy(), mode="lines", name="Drawdown",
|
| 113 |
+
line=dict(color=NEGATIVE, width=2), fill="tozeroy",
|
| 114 |
+
fillcolor="rgba(230,103,103,0.18)",
|
| 115 |
+
hovertemplate="Drawdown %{y:.1f}%<extra></extra>",
|
| 116 |
+
)
|
| 117 |
+
)
|
| 118 |
+
trough = float(dd.min())
|
| 119 |
+
fig.add_annotation(
|
| 120 |
+
x=dd.idxmin(), y=trough, text=f"worst {trough:.1f}%", showarrow=True,
|
| 121 |
+
arrowhead=0, arrowcolor=AXIS, ay=24, font=dict(color=INK_SECONDARY, size=11),
|
| 122 |
+
)
|
| 123 |
+
fig.update_layout(**_base_layout("Drawdown", height=240, showlegend=False))
|
| 124 |
+
fig.update_yaxes(ticksuffix="%")
|
| 125 |
+
return fig
|
| 126 |
+
|
| 127 |
+
|
| 128 |
+
def permutation_chart(report) -> go.Figure:
|
| 129 |
+
"""The headline chart: your Sharpe against Sharpes from shuffled markets."""
|
| 130 |
+
perm = report.permutation
|
| 131 |
+
if perm is None or perm.null.size == 0:
|
| 132 |
+
return empty_figure("Permutation test was skipped.", height=320)
|
| 133 |
+
|
| 134 |
+
null = perm.null
|
| 135 |
+
fig = go.Figure()
|
| 136 |
+
fig.add_trace(
|
| 137 |
+
go.Histogram(
|
| 138 |
+
x=null, name="Shuffled markets (no real edge)", nbinsx=44,
|
| 139 |
+
marker=dict(color="rgba(217,89,38,0.55)", line=dict(color=REFERENCE, width=1)),
|
| 140 |
+
hovertemplate="Sharpe %{x:.2f}<br>%{y} shuffles<extra></extra>",
|
| 141 |
+
)
|
| 142 |
+
)
|
| 143 |
+
|
| 144 |
+
top = np.histogram(null, bins=44)[0].max() if null.size else 1
|
| 145 |
+
fig.add_trace(
|
| 146 |
+
go.Scatter(
|
| 147 |
+
x=[perm.observed, perm.observed], y=[0, top * 1.08], mode="lines",
|
| 148 |
+
name="Your strategy", line=dict(color=SUBJECT, width=2),
|
| 149 |
+
hovertemplate="Your Sharpe %{x:.2f}<extra></extra>",
|
| 150 |
+
)
|
| 151 |
+
)
|
| 152 |
+
fig.add_annotation(
|
| 153 |
+
x=perm.observed, y=top * 1.08, text=f" your Sharpe {perm.observed:.2f}",
|
| 154 |
+
showarrow=False, xanchor="left", font=dict(color=SUBJECT, size=11),
|
| 155 |
+
)
|
| 156 |
+
|
| 157 |
+
beats = (null >= perm.observed).mean() * 100.0
|
| 158 |
+
fig.update_layout(
|
| 159 |
+
**_base_layout(
|
| 160 |
+
f"Permutation test — {beats:.0f}% of structure-free markets did this well or better "
|
| 161 |
+
f"(p = {perm.p_value:.3f})",
|
| 162 |
+
height=320,
|
| 163 |
+
)
|
| 164 |
+
)
|
| 165 |
+
fig.update_layout(hovermode="closest", bargap=0.02)
|
| 166 |
+
fig.update_xaxes(title=dict(text="Annualised Sharpe ratio", font=dict(color=INK_MUTED, size=11)))
|
| 167 |
+
# Headroom so the "your Sharpe" label never collides with the plot edge.
|
| 168 |
+
fig.update_yaxes(
|
| 169 |
+
title=dict(text="Shuffled markets", font=dict(color=INK_MUTED, size=11)),
|
| 170 |
+
range=[0, top * 1.28],
|
| 171 |
+
)
|
| 172 |
+
return fig
|
| 173 |
+
|
| 174 |
+
|
| 175 |
+
def walkforward_chart(report) -> go.Figure:
|
| 176 |
+
"""In-sample vs out-of-sample Sharpe, fold by fold."""
|
| 177 |
+
folds = report.walkforward.get("folds") or []
|
| 178 |
+
if not folds:
|
| 179 |
+
return empty_figure(report.walkforward.get("note") or _EMPTY_NOTE, height=300)
|
| 180 |
+
|
| 181 |
+
labels = [f"Fold {f['fold']}<br><span style='font-size:10px'>{f['test_start'][:7]}</span>" for f in folds]
|
| 182 |
+
fig = go.Figure()
|
| 183 |
+
fig.add_trace(
|
| 184 |
+
go.Bar(
|
| 185 |
+
x=labels, y=[f["is_sharpe"] for f in folds], name="In-sample (tuned)",
|
| 186 |
+
marker=dict(color=REFERENCE, line=dict(color=SURFACE, width=2)),
|
| 187 |
+
hovertemplate="In-sample Sharpe %{y:.2f}<extra></extra>",
|
| 188 |
+
)
|
| 189 |
+
)
|
| 190 |
+
fig.add_trace(
|
| 191 |
+
go.Bar(
|
| 192 |
+
x=labels, y=[f["oos_sharpe"] for f in folds], name="Out-of-sample (blind)",
|
| 193 |
+
marker=dict(color=SUBJECT, line=dict(color=SURFACE, width=2)),
|
| 194 |
+
hovertemplate="Out-of-sample Sharpe %{y:.2f}<extra></extra>",
|
| 195 |
+
)
|
| 196 |
+
)
|
| 197 |
+
eff = report.walkforward.get("efficiency", 0.0)
|
| 198 |
+
fig.update_layout(
|
| 199 |
+
**_base_layout(f"Walk-forward — {eff:.0%} of the tuned Sharpe survived out of sample", height=300)
|
| 200 |
+
)
|
| 201 |
+
fig.update_layout(barmode="group", bargap=0.35, bargroupgap=0.08, hovermode="x unified")
|
| 202 |
+
fig.add_hline(y=0, line=dict(color=AXIS, width=1))
|
| 203 |
+
return fig
|
| 204 |
+
|
| 205 |
+
|
| 206 |
+
def score_chart(verdict: Dict[str, object], significance_label: str = "Beats shuffled markets") -> go.Figure:
|
| 207 |
+
"""The five components behind the Reality Score."""
|
| 208 |
+
components = verdict.get("components") or {}
|
| 209 |
+
if not components:
|
| 210 |
+
return empty_figure(height=260)
|
| 211 |
+
|
| 212 |
+
pretty = {
|
| 213 |
+
"significance": significance_label,
|
| 214 |
+
"selection": "Survives selection bias",
|
| 215 |
+
"walk_forward": "Holds up walking forward",
|
| 216 |
+
"overfitting": "Not overfit (PBO)",
|
| 217 |
+
"robustness": "Survives 3x costs",
|
| 218 |
+
}
|
| 219 |
+
keys = list(pretty)
|
| 220 |
+
values = [float(components.get(k, 0.0)) for k in keys]
|
| 221 |
+
|
| 222 |
+
fig = go.Figure(
|
| 223 |
+
go.Bar(
|
| 224 |
+
x=values, y=[pretty[k] for k in keys], orientation="h",
|
| 225 |
+
marker=dict(color=SUBJECT, line=dict(color=SURFACE, width=2)),
|
| 226 |
+
text=[f"{v:.0f}" for v in values], textposition="outside",
|
| 227 |
+
textfont=dict(color=INK_SECONDARY, size=11),
|
| 228 |
+
hovertemplate="%{y}: %{x:.0f}/100<extra></extra>",
|
| 229 |
+
)
|
| 230 |
+
)
|
| 231 |
+
fig.update_layout(**_base_layout("Where the score comes from", height=260, showlegend=False))
|
| 232 |
+
fig.update_layout(margin=dict(l=190, r=48, t=48, b=32), hovermode="closest")
|
| 233 |
+
fig.update_xaxes(range=[0, 108], tickvals=[0, 25, 50, 75, 100])
|
| 234 |
+
fig.update_yaxes(autorange="reversed")
|
| 235 |
+
return fig
|
| 236 |
+
|
| 237 |
+
|
| 238 |
+
def arena_chart(table: pd.DataFrame) -> go.Figure:
|
| 239 |
+
"""Leaderboard bars. One measure, one colour — the table carries the rest."""
|
| 240 |
+
if table is None or table.empty:
|
| 241 |
+
return empty_figure(height=380)
|
| 242 |
+
|
| 243 |
+
ordered = table.iloc[::-1]
|
| 244 |
+
fig = go.Figure(
|
| 245 |
+
go.Bar(
|
| 246 |
+
x=ordered["Sharpe"].to_numpy(), y=ordered["Strategy"].tolist(), orientation="h",
|
| 247 |
+
marker=dict(color=SUBJECT, line=dict(color=SURFACE, width=2)),
|
| 248 |
+
customdata=np.column_stack([ordered["p-value"].to_numpy(), ordered["DSR"].to_numpy()]),
|
| 249 |
+
hovertemplate="%{y}<br>Sharpe %{x:.2f}<br>p = %{customdata[0]:.3f}"
|
| 250 |
+
"<br>Deflated Sharpe %{customdata[1]:.2f}<extra></extra>",
|
| 251 |
+
)
|
| 252 |
+
)
|
| 253 |
+
# Direct-label only what matters: the ones that actually cleared significance.
|
| 254 |
+
for _, row in ordered.iterrows():
|
| 255 |
+
if np.isfinite(row["p-value"]) and row["p-value"] < 0.05:
|
| 256 |
+
fig.add_annotation(
|
| 257 |
+
x=row["Sharpe"], y=row["Strategy"], text=" p < 0.05", showarrow=False,
|
| 258 |
+
xanchor="left" if row["Sharpe"] >= 0 else "right",
|
| 259 |
+
font=dict(color=INK_SECONDARY, size=10),
|
| 260 |
+
)
|
| 261 |
+
fig.update_layout(
|
| 262 |
+
**_base_layout(
|
| 263 |
+
"Strategy arena — Sharpe ratio, ordered by strength of evidence",
|
| 264 |
+
height=max(300, 42 * len(table)),
|
| 265 |
+
showlegend=False,
|
| 266 |
+
)
|
| 267 |
+
)
|
| 268 |
+
fig.update_layout(margin=dict(l=180, r=96, t=48, b=32), hovermode="closest")
|
| 269 |
+
fig.add_vline(x=0, line=dict(color=AXIS, width=1))
|
| 270 |
+
return fig
|
| 271 |
+
|
| 272 |
+
|
| 273 |
+
def cross_permutation_chart(report) -> go.Figure:
|
| 274 |
+
"""Sharpe against books of identical shape holding randomly chosen names."""
|
| 275 |
+
perm = report.permutation
|
| 276 |
+
if perm is None or perm.null.size == 0:
|
| 277 |
+
return empty_figure("Name-shuffle test was skipped.", height=320)
|
| 278 |
+
|
| 279 |
+
fig = go.Figure()
|
| 280 |
+
fig.add_trace(
|
| 281 |
+
go.Histogram(
|
| 282 |
+
x=perm.null, name="Same book, random names", nbinsx=40,
|
| 283 |
+
marker=dict(color="rgba(217,89,38,0.55)", line=dict(color=REFERENCE, width=1)),
|
| 284 |
+
hovertemplate="Sharpe %{x:.2f}<br>%{y} shuffles<extra></extra>",
|
| 285 |
+
)
|
| 286 |
+
)
|
| 287 |
+
top = np.histogram(perm.null, bins=40)[0].max() if perm.null.size else 1
|
| 288 |
+
fig.add_trace(
|
| 289 |
+
go.Scatter(
|
| 290 |
+
x=[perm.observed, perm.observed], y=[0, top * 1.08], mode="lines",
|
| 291 |
+
name="Your book", line=dict(color=SUBJECT, width=2),
|
| 292 |
+
hovertemplate="Your Sharpe %{x:.2f}<extra></extra>",
|
| 293 |
+
)
|
| 294 |
+
)
|
| 295 |
+
fig.add_annotation(
|
| 296 |
+
x=perm.observed, y=top * 1.08, text=f" your Sharpe {perm.observed:.2f}",
|
| 297 |
+
showarrow=False, xanchor="left", font=dict(color=SUBJECT, size=11),
|
| 298 |
+
)
|
| 299 |
+
beats = (perm.null >= perm.observed).mean() * 100.0
|
| 300 |
+
fig.update_layout(
|
| 301 |
+
**_base_layout(
|
| 302 |
+
f"Name-shuffle test — {beats:.0f}% of books with the same shape but random names "
|
| 303 |
+
f"did this well or better (p = {perm.p_value:.3f})",
|
| 304 |
+
height=320,
|
| 305 |
+
)
|
| 306 |
+
)
|
| 307 |
+
fig.update_layout(hovermode="closest", bargap=0.02)
|
| 308 |
+
fig.update_xaxes(title=dict(text="Annualised Sharpe ratio", font=dict(color=INK_MUTED, size=11)))
|
| 309 |
+
fig.update_yaxes(
|
| 310 |
+
title=dict(text="Shuffled books", font=dict(color=INK_MUTED, size=11)),
|
| 311 |
+
range=[0, top * 1.28],
|
| 312 |
+
)
|
| 313 |
+
return fig
|
| 314 |
+
|
| 315 |
+
|
| 316 |
+
def attribution_chart(attribution: Dict[str, object]) -> go.Figure:
|
| 317 |
+
"""Factor betas. One measure across categories, so one colour."""
|
| 318 |
+
if not attribution or not attribution.get("available"):
|
| 319 |
+
return empty_figure((attribution or {}).get("note", _EMPTY_NOTE), height=280)
|
| 320 |
+
|
| 321 |
+
betas = attribution.get("betas") or {}
|
| 322 |
+
if not betas:
|
| 323 |
+
return empty_figure("No factor exposures to show.", height=280)
|
| 324 |
+
|
| 325 |
+
names = list(betas)
|
| 326 |
+
values = [betas[n] for n in names]
|
| 327 |
+
fig = go.Figure(
|
| 328 |
+
go.Bar(
|
| 329 |
+
x=values, y=[n.replace("_", " ") for n in names], orientation="h",
|
| 330 |
+
marker=dict(color=SUBJECT, line=dict(color=SURFACE, width=2)),
|
| 331 |
+
text=[f"{v:+.2f}" for v in values], textposition="outside",
|
| 332 |
+
textfont=dict(color=INK_SECONDARY, size=11),
|
| 333 |
+
hovertemplate="%{y} beta %{x:.2f}<extra></extra>",
|
| 334 |
+
)
|
| 335 |
+
)
|
| 336 |
+
alpha = attribution.get("alpha_annual", 0.0)
|
| 337 |
+
t_stat = attribution.get("alpha_t_stat", 0.0)
|
| 338 |
+
fig.update_layout(
|
| 339 |
+
**_base_layout(
|
| 340 |
+
f"Style exposure — alpha {alpha:+.1%}/yr (t = {t_stat:.1f}), "
|
| 341 |
+
f"R² {attribution.get('r_squared', 0):.0%}",
|
| 342 |
+
height=280,
|
| 343 |
+
showlegend=False,
|
| 344 |
+
)
|
| 345 |
+
)
|
| 346 |
+
fig.update_layout(margin=dict(l=120, r=88, t=48, b=32), hovermode="closest")
|
| 347 |
+
# Outside labels need room or the widest beta reads as "+0".
|
| 348 |
+
span = max(abs(min(values)), abs(max(values)), 0.1)
|
| 349 |
+
fig.update_xaxes(range=[min(0, min(values)) - 0.25 * span, max(0, max(values)) + 0.35 * span])
|
| 350 |
+
fig.add_vline(x=0, line=dict(color=AXIS, width=1))
|
| 351 |
+
return fig
|
| 352 |
+
|
| 353 |
+
|
| 354 |
+
def weights_chart(report) -> go.Figure:
|
| 355 |
+
"""Gross and net exposure over time — is the book actually neutral?"""
|
| 356 |
+
held = report.backtest.held
|
| 357 |
+
gross = held.abs().sum(axis=1)
|
| 358 |
+
net = held.sum(axis=1)
|
| 359 |
+
|
| 360 |
+
fig = go.Figure()
|
| 361 |
+
fig.add_trace(
|
| 362 |
+
go.Scatter(
|
| 363 |
+
x=gross.index, y=gross.to_numpy(), name="Gross", mode="lines",
|
| 364 |
+
line=dict(color=REFERENCE, width=2, dash="dash"),
|
| 365 |
+
hovertemplate="Gross %{y:.2f}x<extra></extra>",
|
| 366 |
+
)
|
| 367 |
+
)
|
| 368 |
+
fig.add_trace(
|
| 369 |
+
go.Scatter(
|
| 370 |
+
x=net.index, y=net.to_numpy(), name="Net", mode="lines",
|
| 371 |
+
line=dict(color=SUBJECT, width=2),
|
| 372 |
+
hovertemplate="Net %{y:.2f}x<extra></extra>",
|
| 373 |
+
)
|
| 374 |
+
)
|
| 375 |
+
fig.update_layout(**_base_layout("Book exposure", height=240))
|
| 376 |
+
fig.add_hline(y=0, line=dict(color=AXIS, width=1))
|
| 377 |
+
return fig
|
| 378 |
+
|
| 379 |
+
|
| 380 |
+
def exposure_chart(report) -> go.Figure:
|
| 381 |
+
"""What the strategy was actually holding, over time."""
|
| 382 |
+
pos = report.backtest.position
|
| 383 |
+
fig = go.Figure(
|
| 384 |
+
go.Scatter(
|
| 385 |
+
x=pos.index, y=pos.to_numpy(), mode="lines", name="Exposure",
|
| 386 |
+
line=dict(color=SUBJECT, width=2, shape="hv"), fill="tozeroy",
|
| 387 |
+
fillcolor="rgba(57,135,229,0.16)",
|
| 388 |
+
hovertemplate="Exposure %{y:.2f}x<extra></extra>",
|
| 389 |
+
)
|
| 390 |
+
)
|
| 391 |
+
fig.update_layout(**_base_layout("Position held", height=200, showlegend=False))
|
| 392 |
+
fig.add_hline(y=0, line=dict(color=AXIS, width=1))
|
| 393 |
+
return fig
|
algotrader/cli.py
ADDED
|
@@ -0,0 +1,290 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Command-line interface.
|
| 2 |
+
|
| 3 |
+
python -m algotrader.cli lab --symbol SPY --strategy sma_cross
|
| 4 |
+
python -m algotrader.cli arena --symbol BTC-USD --start 2018-01-01
|
| 5 |
+
python -m algotrader.cli strategies
|
| 6 |
+
"""
|
| 7 |
+
|
| 8 |
+
from __future__ import annotations
|
| 9 |
+
|
| 10 |
+
import argparse
|
| 11 |
+
import json
|
| 12 |
+
import sys
|
| 13 |
+
from typing import Dict, List
|
| 14 |
+
|
| 15 |
+
from . import __version__
|
| 16 |
+
from .lab import LabConfig, run_arena, run_lab
|
| 17 |
+
from .strategies import REGISTRY, get_strategy
|
| 18 |
+
|
| 19 |
+
|
| 20 |
+
def _parse_params(pairs: List[str] | None) -> Dict[str, float]:
|
| 21 |
+
params: Dict[str, float] = {}
|
| 22 |
+
for pair in pairs or []:
|
| 23 |
+
if "=" not in pair:
|
| 24 |
+
raise SystemExit(f"--param expects name=value, got '{pair}'")
|
| 25 |
+
name, _, value = pair.partition("=")
|
| 26 |
+
params[name.strip()] = float(value)
|
| 27 |
+
return params
|
| 28 |
+
|
| 29 |
+
|
| 30 |
+
def _add_common(parser: argparse.ArgumentParser) -> None:
|
| 31 |
+
parser.add_argument("--symbol", default="SPY")
|
| 32 |
+
parser.add_argument("--start", default="2015-01-01")
|
| 33 |
+
parser.add_argument("--end", default=None)
|
| 34 |
+
parser.add_argument("--interval", default="1d")
|
| 35 |
+
parser.add_argument(
|
| 36 |
+
"--source", default="yahoo", choices=["yahoo", "live", "auto", "cache", "synthetic"],
|
| 37 |
+
help="'yahoo' requires a Yahoo download. 'synthetic' is offline tests only. 'auto' falls back to the simulator.",
|
| 38 |
+
)
|
| 39 |
+
parser.add_argument("--commission-bps", type=float, default=1.0)
|
| 40 |
+
parser.add_argument("--slippage-bps", type=float, default=2.0)
|
| 41 |
+
parser.add_argument("--no-short", action="store_true", help="Long/flat only.")
|
| 42 |
+
|
| 43 |
+
|
| 44 |
+
def _config_from(args: argparse.Namespace, **overrides) -> LabConfig:
|
| 45 |
+
return LabConfig(
|
| 46 |
+
symbol=args.symbol,
|
| 47 |
+
start=args.start,
|
| 48 |
+
end=args.end,
|
| 49 |
+
interval=args.interval,
|
| 50 |
+
source=args.source,
|
| 51 |
+
commission_bps=args.commission_bps,
|
| 52 |
+
slippage_bps=args.slippage_bps,
|
| 53 |
+
allow_short=not args.no_short,
|
| 54 |
+
**overrides,
|
| 55 |
+
)
|
| 56 |
+
|
| 57 |
+
|
| 58 |
+
def _cmd_lab(args: argparse.Namespace) -> int:
|
| 59 |
+
cfg = _config_from(
|
| 60 |
+
args,
|
| 61 |
+
strategy=args.strategy,
|
| 62 |
+
params=_parse_params(args.param),
|
| 63 |
+
n_permutations=args.permutations,
|
| 64 |
+
permutation_method=args.null,
|
| 65 |
+
wf_folds=args.folds,
|
| 66 |
+
)
|
| 67 |
+
progress = None if args.quiet else (lambda f, m: print(f" [{f:5.0%}] {m}", file=sys.stderr))
|
| 68 |
+
report = run_lab(cfg, progress=progress)
|
| 69 |
+
|
| 70 |
+
if args.json:
|
| 71 |
+
payload = {
|
| 72 |
+
"symbol": report.market.symbol,
|
| 73 |
+
"source": report.market.source,
|
| 74 |
+
"strategy": report.strategy.key,
|
| 75 |
+
"params": report.params,
|
| 76 |
+
"metrics": report.backtest.metrics,
|
| 77 |
+
"benchmark_metrics": report.backtest.benchmark_metrics,
|
| 78 |
+
"p_value": report.permutation.p_value if report.permutation else None,
|
| 79 |
+
"deflated_sharpe": report.dsr.get("dsr"),
|
| 80 |
+
"pbo": report.pbo.get("pbo"),
|
| 81 |
+
"walkforward_efficiency": report.walkforward.get("efficiency"),
|
| 82 |
+
"cost_stress": report.cost_stress,
|
| 83 |
+
"verdict": {k: v for k, v in report.verdict.items()},
|
| 84 |
+
}
|
| 85 |
+
print(json.dumps(payload, indent=2, default=str))
|
| 86 |
+
return 0
|
| 87 |
+
|
| 88 |
+
v, m, b = report.verdict, report.backtest.metrics, report.backtest.benchmark_metrics
|
| 89 |
+
bar = "=" * 66
|
| 90 |
+
print(f"\n{bar}")
|
| 91 |
+
print(f" {report.strategy.name} on {report.market.symbol} [{report.market.source} data]")
|
| 92 |
+
print(f" {report.market.start.date()} to {report.market.end.date()} · {len(report.market.df):,} bars")
|
| 93 |
+
print(bar)
|
| 94 |
+
print(f" REALITY SCORE {v['score']:.1f} / 100 GRADE {v['grade']}")
|
| 95 |
+
print(f" {v['headline']}")
|
| 96 |
+
print(bar)
|
| 97 |
+
print(f" Total return {m['total_return']:>9.1%} buy & hold {b['total_return']:>8.1%}")
|
| 98 |
+
print(f" CAGR {m['cagr']:>9.1%} buy & hold {b['cagr']:>8.1%}")
|
| 99 |
+
print(f" Sharpe {m['sharpe']:>9.2f} buy & hold {b['sharpe']:>8.2f}")
|
| 100 |
+
print(f" Max drawdown {m['max_drawdown']:>9.1%}")
|
| 101 |
+
print(f" Trades {int(m.get('n_trades', 0)):>9,}")
|
| 102 |
+
print(bar)
|
| 103 |
+
if report.permutation:
|
| 104 |
+
print(f" Permutation p {report.permutation.p_value:>9.3f} ({report.permutation.n_permutations} shuffled markets)")
|
| 105 |
+
print(f" Deflated Sharpe {report.dsr.get('dsr', 0):>9.2f} (after {report.trials.get('n', 1)} variants)")
|
| 106 |
+
pbo = report.pbo.get("pbo")
|
| 107 |
+
print(f" Overfit prob. {pbo:>9.2f}" if pbo == pbo else " Overfit prob. n/a")
|
| 108 |
+
print(f" Walk-forward eff. {report.walkforward.get('efficiency', 0):>9.2f}")
|
| 109 |
+
print(f" Sharpe at 3x cost {report.cost_stress.get('sharpe_3x', 0):>9.2f}")
|
| 110 |
+
print(bar)
|
| 111 |
+
for flag in v["flags"]:
|
| 112 |
+
print(f" ! {flag}")
|
| 113 |
+
if v["flags"]:
|
| 114 |
+
print(bar)
|
| 115 |
+
print(f" {v['verdict']}\n")
|
| 116 |
+
return 0
|
| 117 |
+
|
| 118 |
+
|
| 119 |
+
def _cmd_arena(args: argparse.Namespace) -> int:
|
| 120 |
+
cfg = _config_from(args)
|
| 121 |
+
progress = None if args.quiet else (lambda f, m: print(f" [{f:5.0%}] {m}", file=sys.stderr))
|
| 122 |
+
table, market, _ = run_arena(cfg, n_permutations=args.permutations, progress=progress)
|
| 123 |
+
|
| 124 |
+
if args.json:
|
| 125 |
+
print(table.to_json(orient="records", indent=2))
|
| 126 |
+
return 0
|
| 127 |
+
|
| 128 |
+
print(f"\n {market.symbol} [{market.source} data] "
|
| 129 |
+
f"{market.start.date()} to {market.end.date()}\n")
|
| 130 |
+
display = table.drop(columns=["key"]).copy()
|
| 131 |
+
for col in ("Return", "CAGR", "MaxDD"):
|
| 132 |
+
display[col] = display[col].map("{:.1%}".format)
|
| 133 |
+
for col in ("Sharpe", "DSR", "Evidence"):
|
| 134 |
+
display[col] = display[col].map("{:.2f}".format)
|
| 135 |
+
display["p-value"] = display["p-value"].map(lambda v: "—" if v != v else f"{v:.3f}")
|
| 136 |
+
print(display.to_string(index=False))
|
| 137 |
+
print("\n Ranked by evidence = (1 - p) x deflated Sharpe, not by return.\n")
|
| 138 |
+
return 0
|
| 139 |
+
|
| 140 |
+
|
| 141 |
+
def _cmd_portfolio(args: argparse.Namespace) -> int:
|
| 142 |
+
from .portfolio_lab import DEFAULT_UNIVERSE, PortfolioLabConfig, run_portfolio_lab
|
| 143 |
+
|
| 144 |
+
symbols = [s.strip() for s in args.symbols.split(",") if s.strip()] if args.symbols else DEFAULT_UNIVERSE
|
| 145 |
+
cfg = PortfolioLabConfig(
|
| 146 |
+
symbols=symbols,
|
| 147 |
+
start=args.start,
|
| 148 |
+
end=args.end,
|
| 149 |
+
interval=args.interval,
|
| 150 |
+
source=args.source,
|
| 151 |
+
strategy=args.strategy,
|
| 152 |
+
params=_parse_params(args.param),
|
| 153 |
+
commission_bps=args.commission_bps,
|
| 154 |
+
slippage_bps=args.slippage_bps,
|
| 155 |
+
allow_short=not args.no_short,
|
| 156 |
+
rebalance=args.rebalance,
|
| 157 |
+
n_permutations=args.permutations,
|
| 158 |
+
wf_folds=args.folds,
|
| 159 |
+
)
|
| 160 |
+
progress = None if args.quiet else (lambda f, m: print(f" [{f:5.0%}] {m}", file=sys.stderr))
|
| 161 |
+
report = run_portfolio_lab(cfg, progress=progress)
|
| 162 |
+
|
| 163 |
+
if args.json:
|
| 164 |
+
print(json.dumps({
|
| 165 |
+
"symbols": report.panel.symbols,
|
| 166 |
+
"strategy": report.strategy.key,
|
| 167 |
+
"params": report.params,
|
| 168 |
+
"metrics": report.backtest.metrics,
|
| 169 |
+
"p_value": report.permutation.p_value if report.permutation else None,
|
| 170 |
+
"deflated_sharpe": report.dsr.get("dsr"),
|
| 171 |
+
"pbo": report.pbo.get("pbo"),
|
| 172 |
+
"walkforward_efficiency": report.walkforward.get("efficiency"),
|
| 173 |
+
"attribution": report.attribution,
|
| 174 |
+
"survivorship": report.survivorship.__dict__,
|
| 175 |
+
"verdict": dict(report.verdict),
|
| 176 |
+
}, indent=2, default=str))
|
| 177 |
+
return 0
|
| 178 |
+
|
| 179 |
+
v, m, b = report.verdict, report.backtest.metrics, report.backtest.benchmark_metrics
|
| 180 |
+
bar = "=" * 72
|
| 181 |
+
print(f"\n{bar}")
|
| 182 |
+
print(f" {report.strategy.name} on {len(report.panel.symbols)} symbols [{report.panel.interval}]")
|
| 183 |
+
print(f" {report.panel.index[0].date()} to {report.panel.index[-1].date()} · "
|
| 184 |
+
f"{len(report.panel):,} bars · rebalance {report.config.rebalance}")
|
| 185 |
+
print(bar)
|
| 186 |
+
print(f" REALITY SCORE {v['score']:.1f} / 100 GRADE {v['grade']}")
|
| 187 |
+
print(f" {v['headline']}")
|
| 188 |
+
print(bar)
|
| 189 |
+
print(f" Total return {m['total_return']:>9.1%} equal weight {b['total_return']:>8.1%}")
|
| 190 |
+
print(f" CAGR {m['cagr']:>9.1%} equal weight {b['cagr']:>8.1%}")
|
| 191 |
+
print(f" Sharpe {m['sharpe']:>9.2f} equal weight {b['sharpe']:>8.2f}")
|
| 192 |
+
print(f" Max drawdown {m['max_drawdown']:>9.1%}")
|
| 193 |
+
print(f" Gross / net exp. {m.get('gross_exposure', 0):>9.2f} / {m.get('net_exposure', 0):.2f}")
|
| 194 |
+
print(f" Avg positions {m.get('avg_positions', 0):>9.1f} turnover {m.get('turnover_ann', 0):.1f}x/yr")
|
| 195 |
+
print(bar)
|
| 196 |
+
if report.permutation:
|
| 197 |
+
print(f" Name-shuffle p {report.permutation.p_value:>9.3f} "
|
| 198 |
+
f"({report.permutation.n_permutations} shuffles of which names got which weights)")
|
| 199 |
+
print(f" Deflated Sharpe {report.dsr.get('dsr', 0):>9.2f} (after {report.trials.get('n', 1)} variants)")
|
| 200 |
+
pbo = report.pbo.get("pbo")
|
| 201 |
+
print(f" Overfit prob. {pbo:>9.2f}" if pbo == pbo else " Overfit prob. n/a")
|
| 202 |
+
print(f" Walk-forward eff. {report.walkforward.get('efficiency', 0):>9.2f}")
|
| 203 |
+
if report.attribution.get("available"):
|
| 204 |
+
print(f" Style alpha {report.attribution['alpha_annual']:>9.1%} "
|
| 205 |
+
f"t = {report.attribution['alpha_t_stat']:.2f}, R² = {report.attribution['r_squared']:.2f}")
|
| 206 |
+
print(f" Survivorship {report.survivorship.survival_rate:>9.0%} "
|
| 207 |
+
f"({report.survivorship.n_delisted} of {report.survivorship.n_symbols} stopped trading)")
|
| 208 |
+
print(bar)
|
| 209 |
+
for flag in v["flags"]:
|
| 210 |
+
print(f" ! {flag}")
|
| 211 |
+
if v["flags"]:
|
| 212 |
+
print(bar)
|
| 213 |
+
print(f" {v['verdict']}\n")
|
| 214 |
+
return 0
|
| 215 |
+
|
| 216 |
+
|
| 217 |
+
def _cmd_strategies(args: argparse.Namespace) -> int:
|
| 218 |
+
from .cross_sectional import XS_REGISTRY
|
| 219 |
+
|
| 220 |
+
for title, registry in (("Single asset", REGISTRY), ("Cross-sectional", XS_REGISTRY)):
|
| 221 |
+
print(f"\n {title}\n {'-' * len(title)}")
|
| 222 |
+
for key, strategy in registry.items():
|
| 223 |
+
params = ", ".join(f"{p.name}={p.default:g}" for p in strategy.params) or "no parameters"
|
| 224 |
+
print(f" {key:<22} {strategy.name:<28} [{strategy.family}]")
|
| 225 |
+
print(f" {'':<22} {strategy.description}")
|
| 226 |
+
print(f" {'':<22} defaults: {params}\n")
|
| 227 |
+
return 0
|
| 228 |
+
|
| 229 |
+
|
| 230 |
+
def main(argv: List[str] | None = None) -> int:
|
| 231 |
+
parser = argparse.ArgumentParser(
|
| 232 |
+
prog="algotrader",
|
| 233 |
+
description="Backtest a trading rule, then try to prove the result was luck.",
|
| 234 |
+
)
|
| 235 |
+
parser.add_argument("--version", action="version", version=f"algotrader {__version__}")
|
| 236 |
+
sub = parser.add_subparsers(dest="command", required=True)
|
| 237 |
+
|
| 238 |
+
lab = sub.add_parser("lab", help="Full reality check for one strategy.")
|
| 239 |
+
_add_common(lab)
|
| 240 |
+
lab.add_argument("--strategy", default="sma_cross", choices=sorted(REGISTRY))
|
| 241 |
+
lab.add_argument("--param", action="append", metavar="NAME=VALUE",
|
| 242 |
+
help="Override a strategy parameter. Repeatable.")
|
| 243 |
+
lab.add_argument("--permutations", type=int, default=250)
|
| 244 |
+
lab.add_argument("--null", default="permute", choices=["permute", "block"])
|
| 245 |
+
lab.add_argument("--folds", type=int, default=5)
|
| 246 |
+
lab.add_argument("--json", action="store_true")
|
| 247 |
+
lab.add_argument("--quiet", "-q", action="store_true")
|
| 248 |
+
lab.set_defaults(func=_cmd_lab)
|
| 249 |
+
|
| 250 |
+
arena = sub.add_parser("arena", help="Race every strategy on one market.")
|
| 251 |
+
_add_common(arena)
|
| 252 |
+
arena.add_argument("--permutations", type=int, default=120)
|
| 253 |
+
arena.add_argument("--json", action="store_true")
|
| 254 |
+
arena.add_argument("--quiet", "-q", action="store_true")
|
| 255 |
+
arena.set_defaults(func=_cmd_arena)
|
| 256 |
+
|
| 257 |
+
from .cross_sectional import XS_REGISTRY
|
| 258 |
+
|
| 259 |
+
portfolio = sub.add_parser(
|
| 260 |
+
"portfolio", help="Reality check for a cross-sectional (multi-asset) strategy."
|
| 261 |
+
)
|
| 262 |
+
_add_common(portfolio)
|
| 263 |
+
portfolio.add_argument(
|
| 264 |
+
"--symbols", default=None,
|
| 265 |
+
help="Comma-separated universe, e.g. SPY,QQQ,AAPL. Defaults to a 12-name universe.",
|
| 266 |
+
)
|
| 267 |
+
portfolio.add_argument("--strategy", default="xs_momentum", choices=sorted(XS_REGISTRY))
|
| 268 |
+
portfolio.add_argument("--param", action="append", metavar="NAME=VALUE")
|
| 269 |
+
portfolio.add_argument("--rebalance", default="M", help="D, W, M, Q, or a number of bars.")
|
| 270 |
+
portfolio.add_argument("--permutations", type=int, default=150)
|
| 271 |
+
portfolio.add_argument("--folds", type=int, default=4)
|
| 272 |
+
portfolio.add_argument("--json", action="store_true")
|
| 273 |
+
portfolio.add_argument("--quiet", "-q", action="store_true")
|
| 274 |
+
portfolio.set_defaults(func=_cmd_portfolio)
|
| 275 |
+
|
| 276 |
+
listing = sub.add_parser("strategies", help="List the strategy zoo.")
|
| 277 |
+
listing.set_defaults(func=_cmd_strategies)
|
| 278 |
+
|
| 279 |
+
args = parser.parse_args(argv)
|
| 280 |
+
try:
|
| 281 |
+
return args.func(args)
|
| 282 |
+
except KeyboardInterrupt:
|
| 283 |
+
return 130
|
| 284 |
+
except Exception as exc: # noqa: BLE001
|
| 285 |
+
print(f"error: {exc}", file=sys.stderr)
|
| 286 |
+
return 1
|
| 287 |
+
|
| 288 |
+
|
| 289 |
+
if __name__ == "__main__":
|
| 290 |
+
raise SystemExit(main())
|
algotrader/cross_sectional.py
ADDED
|
@@ -0,0 +1,290 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Cross-sectional strategies.
|
| 2 |
+
|
| 3 |
+
Where a single-asset rule asks "should I be long this thing?", a
|
| 4 |
+
cross-sectional rule asks "which of these things should I be long, and which
|
| 5 |
+
short?". That difference matters for validation: a long-short book that ranks
|
| 6 |
+
names is exposed to entirely different failure modes than a timing rule, and it
|
| 7 |
+
needs its own null (see :mod:`algotrader.validation.cross_permutation`).
|
| 8 |
+
|
| 9 |
+
Each strategy is a pure function ``(panel, **params) -> T x N weights``, causal
|
| 10 |
+
by construction, with gross exposure of at most 1.
|
| 11 |
+
"""
|
| 12 |
+
|
| 13 |
+
from __future__ import annotations
|
| 14 |
+
|
| 15 |
+
import itertools
|
| 16 |
+
from dataclasses import dataclass
|
| 17 |
+
from typing import Callable, Dict, Iterable, List
|
| 18 |
+
|
| 19 |
+
import numpy as np
|
| 20 |
+
import pandas as pd
|
| 21 |
+
|
| 22 |
+
from .panel import Panel
|
| 23 |
+
from .strategies import ParamSpec
|
| 24 |
+
|
| 25 |
+
__all__ = [
|
| 26 |
+
"CrossSectionalStrategy",
|
| 27 |
+
"XS_REGISTRY",
|
| 28 |
+
"get_xs_strategy",
|
| 29 |
+
"list_xs_strategies",
|
| 30 |
+
"scores_to_weights",
|
| 31 |
+
]
|
| 32 |
+
|
| 33 |
+
MIN_NAMES = 4 # below this, "cross-section" is not a meaningful word
|
| 34 |
+
|
| 35 |
+
|
| 36 |
+
def scores_to_weights(
|
| 37 |
+
scores: pd.DataFrame,
|
| 38 |
+
long_frac: float = 0.3,
|
| 39 |
+
short_frac: float = 0.3,
|
| 40 |
+
long_only: bool = False,
|
| 41 |
+
min_names: int = MIN_NAMES,
|
| 42 |
+
) -> pd.DataFrame:
|
| 43 |
+
"""Turn a score matrix into a dollar-neutral (or long-only) weight matrix.
|
| 44 |
+
|
| 45 |
+
Ranks within each date, takes the top and bottom fractions, and equal-weights
|
| 46 |
+
each leg. Gross exposure is 1: half per leg when short-selling, all of it in
|
| 47 |
+
the long leg otherwise.
|
| 48 |
+
"""
|
| 49 |
+
valid = scores.notna()
|
| 50 |
+
counts = valid.sum(axis=1)
|
| 51 |
+
ranks = scores.rank(axis=1, pct=True, na_option="keep")
|
| 52 |
+
|
| 53 |
+
longs = (ranks > 1.0 - long_frac) & valid
|
| 54 |
+
n_long = longs.sum(axis=1).replace(0, np.nan)
|
| 55 |
+
|
| 56 |
+
if long_only:
|
| 57 |
+
weights = longs.astype(float).div(n_long, axis=0)
|
| 58 |
+
else:
|
| 59 |
+
shorts = (ranks <= short_frac) & valid
|
| 60 |
+
n_short = shorts.sum(axis=1).replace(0, np.nan)
|
| 61 |
+
weights = (
|
| 62 |
+
longs.astype(float).div(n_long, axis=0) * 0.5
|
| 63 |
+
- shorts.astype(float).div(n_short, axis=0) * 0.5
|
| 64 |
+
)
|
| 65 |
+
|
| 66 |
+
# A cross-section of two names is not a cross-section.
|
| 67 |
+
weights = weights.where(counts >= min_names, 0.0)
|
| 68 |
+
return weights.fillna(0.0)
|
| 69 |
+
|
| 70 |
+
|
| 71 |
+
def _rebalance_hold(weights: pd.DataFrame, every: int) -> pd.DataFrame:
|
| 72 |
+
"""Refresh the target only every ``every`` bars, holding it in between."""
|
| 73 |
+
if every <= 1:
|
| 74 |
+
return weights
|
| 75 |
+
out = weights.copy()
|
| 76 |
+
keep = np.zeros(len(out), dtype=bool)
|
| 77 |
+
keep[::every] = True
|
| 78 |
+
out.iloc[~keep] = np.nan
|
| 79 |
+
return out.ffill().fillna(0.0)
|
| 80 |
+
|
| 81 |
+
|
| 82 |
+
# --------------------------------------------------------------------------
|
| 83 |
+
# Strategy implementations
|
| 84 |
+
# --------------------------------------------------------------------------
|
| 85 |
+
|
| 86 |
+
def _xs_momentum(
|
| 87 |
+
panel: Panel, lookback: int = 250, skip: int = 20, rebalance: int = 21, long_frac: float = 0.3
|
| 88 |
+
) -> pd.DataFrame:
|
| 89 |
+
"""Classic 12-1 momentum: rank on past return, skipping the most recent month.
|
| 90 |
+
|
| 91 |
+
The skip is not decoration -- including the last month mixes in short-term
|
| 92 |
+
reversal, which points the other way and muddies the signal.
|
| 93 |
+
"""
|
| 94 |
+
close = panel.close
|
| 95 |
+
scores = close.shift(skip) / close.shift(lookback) - 1.0
|
| 96 |
+
return _rebalance_hold(scores_to_weights(scores, long_frac, long_frac), rebalance)
|
| 97 |
+
|
| 98 |
+
|
| 99 |
+
def _xs_reversal(
|
| 100 |
+
panel: Panel, lookback: int = 5, rebalance: int = 5, long_frac: float = 0.3
|
| 101 |
+
) -> pd.DataFrame:
|
| 102 |
+
"""Short-term reversal: buy the recent losers, sell the recent winners."""
|
| 103 |
+
scores = -(panel.close.pct_change(lookback))
|
| 104 |
+
return _rebalance_hold(scores_to_weights(scores, long_frac, long_frac), rebalance)
|
| 105 |
+
|
| 106 |
+
|
| 107 |
+
def _low_volatility(
|
| 108 |
+
panel: Panel, window: int = 60, rebalance: int = 21, long_frac: float = 0.3
|
| 109 |
+
) -> pd.DataFrame:
|
| 110 |
+
"""The low-volatility anomaly: long the calm names, short the wild ones."""
|
| 111 |
+
scores = -(panel.close.pct_change().rolling(window, min_periods=window).std())
|
| 112 |
+
return _rebalance_hold(scores_to_weights(scores, long_frac, long_frac), rebalance)
|
| 113 |
+
|
| 114 |
+
|
| 115 |
+
def _xs_value_proxy(
|
| 116 |
+
panel: Panel, window: int = 250, rebalance: int = 21, long_frac: float = 0.3
|
| 117 |
+
) -> pd.DataFrame:
|
| 118 |
+
"""Distance below the long-run average price, as a crude cheapness proxy.
|
| 119 |
+
|
| 120 |
+
This is not book-to-market -- there are no fundamentals in the panel -- so
|
| 121 |
+
treat it as mean reversion over a long horizon rather than value investing.
|
| 122 |
+
"""
|
| 123 |
+
close = panel.close
|
| 124 |
+
scores = -(close / close.rolling(window, min_periods=window).mean() - 1.0)
|
| 125 |
+
return _rebalance_hold(scores_to_weights(scores, long_frac, long_frac), rebalance)
|
| 126 |
+
|
| 127 |
+
|
| 128 |
+
def _equal_weight(panel: Panel, rebalance: int = 21) -> pd.DataFrame:
|
| 129 |
+
"""Own everything tradable, equally. The bar a stock picker has to clear."""
|
| 130 |
+
listed = panel.close.notna()
|
| 131 |
+
counts = listed.sum(axis=1).replace(0, np.nan)
|
| 132 |
+
return _rebalance_hold(listed.astype(float).div(counts, axis=0).fillna(0.0), rebalance)
|
| 133 |
+
|
| 134 |
+
|
| 135 |
+
def _xs_random(panel: Panel, rebalance: int = 21, seed: int = 7) -> pd.DataFrame:
|
| 136 |
+
"""Random long-short book. The control group for cross-sectional claims."""
|
| 137 |
+
rng = np.random.default_rng(int(seed))
|
| 138 |
+
scores = pd.DataFrame(
|
| 139 |
+
rng.standard_normal(panel.close.shape), index=panel.index, columns=panel.symbols
|
| 140 |
+
).where(panel.close.notna())
|
| 141 |
+
return _rebalance_hold(scores_to_weights(scores), rebalance)
|
| 142 |
+
|
| 143 |
+
|
| 144 |
+
@dataclass(frozen=True)
|
| 145 |
+
class CrossSectionalStrategy:
|
| 146 |
+
key: str
|
| 147 |
+
name: str
|
| 148 |
+
family: str
|
| 149 |
+
description: str
|
| 150 |
+
fn: Callable[..., pd.DataFrame]
|
| 151 |
+
params: tuple = ()
|
| 152 |
+
|
| 153 |
+
def defaults(self) -> Dict[str, float]:
|
| 154 |
+
return {p.name: p.cast(p.default) for p in self.params}
|
| 155 |
+
|
| 156 |
+
def clean(self, params: Dict[str, float] | None) -> Dict[str, float]:
|
| 157 |
+
merged = self.defaults()
|
| 158 |
+
for spec in self.params:
|
| 159 |
+
if params and spec.name in params and params[spec.name] is not None:
|
| 160 |
+
merged[spec.name] = spec.cast(params[spec.name])
|
| 161 |
+
return merged
|
| 162 |
+
|
| 163 |
+
def generate(self, panel: Panel, params: Dict[str, float] | None = None) -> pd.DataFrame:
|
| 164 |
+
weights = self.fn(panel, **self.clean(params))
|
| 165 |
+
return (
|
| 166 |
+
weights.reindex(index=panel.index, columns=panel.symbols)
|
| 167 |
+
.astype(float)
|
| 168 |
+
.fillna(0.0)
|
| 169 |
+
.clip(-1.0, 1.0)
|
| 170 |
+
)
|
| 171 |
+
|
| 172 |
+
def grid(self, limit: int | None = None) -> List[Dict[str, float]]:
|
| 173 |
+
if not self.params:
|
| 174 |
+
return [{}]
|
| 175 |
+
names = [p.name for p in self.params]
|
| 176 |
+
combos = [dict(zip(names, v)) for v in itertools.product(*[p.grid for p in self.params])]
|
| 177 |
+
combos = [c for c in combos if not ("skip" in c and "lookback" in c and c["skip"] >= c["lookback"])]
|
| 178 |
+
if limit is not None and len(combos) > limit:
|
| 179 |
+
step = len(combos) / limit
|
| 180 |
+
combos = [combos[int(i * step)] for i in range(limit)]
|
| 181 |
+
return combos
|
| 182 |
+
|
| 183 |
+
|
| 184 |
+
XS_REGISTRY: Dict[str, CrossSectionalStrategy] = {}
|
| 185 |
+
|
| 186 |
+
|
| 187 |
+
def _register(strategy: CrossSectionalStrategy) -> CrossSectionalStrategy:
|
| 188 |
+
XS_REGISTRY[strategy.key] = strategy
|
| 189 |
+
return strategy
|
| 190 |
+
|
| 191 |
+
|
| 192 |
+
_register(
|
| 193 |
+
CrossSectionalStrategy(
|
| 194 |
+
key="equal_weight",
|
| 195 |
+
name="Equal Weight",
|
| 196 |
+
family="benchmark",
|
| 197 |
+
description="Own every name equally. The bar a stock picker has to clear.",
|
| 198 |
+
fn=_equal_weight,
|
| 199 |
+
params=(ParamSpec("rebalance", "Rebalance (bars)", 21, (5, 21, 63), "int", 1, 252, 1),),
|
| 200 |
+
)
|
| 201 |
+
)
|
| 202 |
+
|
| 203 |
+
_register(
|
| 204 |
+
CrossSectionalStrategy(
|
| 205 |
+
key="xs_momentum",
|
| 206 |
+
name="Cross-Sectional Momentum",
|
| 207 |
+
family="momentum",
|
| 208 |
+
description="Long the past winners, short the past losers, skipping the most recent month.",
|
| 209 |
+
fn=_xs_momentum,
|
| 210 |
+
params=(
|
| 211 |
+
ParamSpec("lookback", "Lookback", 250, (60, 120, 250), "int", 20, 750, 1),
|
| 212 |
+
ParamSpec("skip", "Skip recent", 20, (0, 5, 20), "int", 0, 60, 1),
|
| 213 |
+
ParamSpec("rebalance", "Rebalance (bars)", 21, (5, 21, 63), "int", 1, 252, 1),
|
| 214 |
+
ParamSpec("long_frac", "Leg size", 0.3, (0.1, 0.2, 0.3), "float", 0.05, 0.5, 0.05),
|
| 215 |
+
),
|
| 216 |
+
)
|
| 217 |
+
)
|
| 218 |
+
|
| 219 |
+
_register(
|
| 220 |
+
CrossSectionalStrategy(
|
| 221 |
+
key="xs_reversal",
|
| 222 |
+
name="Short-Term Reversal",
|
| 223 |
+
family="mean-reversion",
|
| 224 |
+
description="Buy this week's losers and sell its winners.",
|
| 225 |
+
fn=_xs_reversal,
|
| 226 |
+
params=(
|
| 227 |
+
ParamSpec("lookback", "Lookback", 5, (1, 3, 5, 10, 21), "int", 1, 60, 1),
|
| 228 |
+
ParamSpec("rebalance", "Rebalance (bars)", 5, (1, 5, 21), "int", 1, 252, 1),
|
| 229 |
+
ParamSpec("long_frac", "Leg size", 0.3, (0.1, 0.2, 0.3), "float", 0.05, 0.5, 0.05),
|
| 230 |
+
),
|
| 231 |
+
)
|
| 232 |
+
)
|
| 233 |
+
|
| 234 |
+
_register(
|
| 235 |
+
CrossSectionalStrategy(
|
| 236 |
+
key="low_volatility",
|
| 237 |
+
name="Low Volatility",
|
| 238 |
+
family="risk",
|
| 239 |
+
description="Long the calm names, short the volatile ones.",
|
| 240 |
+
fn=_low_volatility,
|
| 241 |
+
params=(
|
| 242 |
+
ParamSpec("window", "Vol window", 60, (20, 60, 120), "int", 5, 252, 1),
|
| 243 |
+
ParamSpec("rebalance", "Rebalance (bars)", 21, (5, 21, 63), "int", 1, 252, 1),
|
| 244 |
+
ParamSpec("long_frac", "Leg size", 0.3, (0.1, 0.2, 0.3), "float", 0.05, 0.5, 0.05),
|
| 245 |
+
),
|
| 246 |
+
)
|
| 247 |
+
)
|
| 248 |
+
|
| 249 |
+
_register(
|
| 250 |
+
CrossSectionalStrategy(
|
| 251 |
+
key="xs_value_proxy",
|
| 252 |
+
name="Long-Horizon Reversion",
|
| 253 |
+
family="value-ish",
|
| 254 |
+
description="Long names trading below their long-run average, short those above.",
|
| 255 |
+
fn=_xs_value_proxy,
|
| 256 |
+
params=(
|
| 257 |
+
ParamSpec("window", "Window", 250, (120, 250, 500), "int", 30, 1000, 1),
|
| 258 |
+
ParamSpec("rebalance", "Rebalance (bars)", 21, (5, 21, 63), "int", 1, 252, 1),
|
| 259 |
+
ParamSpec("long_frac", "Leg size", 0.3, (0.1, 0.2, 0.3), "float", 0.05, 0.5, 0.05),
|
| 260 |
+
),
|
| 261 |
+
)
|
| 262 |
+
)
|
| 263 |
+
|
| 264 |
+
_register(
|
| 265 |
+
CrossSectionalStrategy(
|
| 266 |
+
key="xs_random",
|
| 267 |
+
name="Random Book (control)",
|
| 268 |
+
family="control",
|
| 269 |
+
description="Random long-short positions. Anything that cannot beat this is noise.",
|
| 270 |
+
fn=_xs_random,
|
| 271 |
+
params=(
|
| 272 |
+
ParamSpec("rebalance", "Rebalance (bars)", 21, (5, 21, 63), "int", 1, 252, 1),
|
| 273 |
+
ParamSpec("seed", "Seed", 7, (1, 7, 42, 123), "int", 0, 9999, 1),
|
| 274 |
+
),
|
| 275 |
+
)
|
| 276 |
+
)
|
| 277 |
+
|
| 278 |
+
|
| 279 |
+
def get_xs_strategy(key: str) -> CrossSectionalStrategy:
|
| 280 |
+
try:
|
| 281 |
+
return XS_REGISTRY[key]
|
| 282 |
+
except KeyError:
|
| 283 |
+
raise KeyError(
|
| 284 |
+
f"Unknown cross-sectional strategy '{key}'. Available: {', '.join(sorted(XS_REGISTRY))}"
|
| 285 |
+
) from None
|
| 286 |
+
|
| 287 |
+
|
| 288 |
+
def list_xs_strategies(exclude: Iterable[str] = ()) -> List[CrossSectionalStrategy]:
|
| 289 |
+
skip = set(exclude)
|
| 290 |
+
return [s for k, s in XS_REGISTRY.items() if k not in skip]
|
algotrader/data.py
ADDED
|
@@ -0,0 +1,252 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Market data loading with a three-tier fallback.
|
| 2 |
+
|
| 3 |
+
Order of preference: live Yahoo download -> on-disk cache -> a deterministic
|
| 4 |
+
simulator. The fallback exists because a Hugging Face Space that shows a
|
| 5 |
+
stack trace on the first click is a Space nobody shares. When the simulator is
|
| 6 |
+
used, :class:`~algotrader.types.MarketData` says so and the UI shows it.
|
| 7 |
+
"""
|
| 8 |
+
|
| 9 |
+
from __future__ import annotations
|
| 10 |
+
|
| 11 |
+
import hashlib
|
| 12 |
+
import logging
|
| 13 |
+
import os
|
| 14 |
+
from dataclasses import dataclass
|
| 15 |
+
from pathlib import Path
|
| 16 |
+
from typing import Optional
|
| 17 |
+
|
| 18 |
+
import numpy as np
|
| 19 |
+
import pandas as pd
|
| 20 |
+
|
| 21 |
+
from .types import OHLCV_COLUMNS, MarketData
|
| 22 |
+
|
| 23 |
+
logger = logging.getLogger(__name__)
|
| 24 |
+
|
| 25 |
+
CACHE_DIR = Path(os.environ.get("ALGOTRADER_CACHE", Path.home() / ".cache" / "algotrader"))
|
| 26 |
+
NETWORK_ENABLED = os.environ.get("ALGOTRADER_OFFLINE", "").lower() not in ("1", "true", "yes")
|
| 27 |
+
|
| 28 |
+
# Popular tickers get hand-set simulation parameters so the offline demo is at
|
| 29 |
+
# least in the right postcode: annual drift, annual vol, and a starting price.
|
| 30 |
+
@dataclass(frozen=True)
|
| 31 |
+
class SimProfile:
|
| 32 |
+
drift: float
|
| 33 |
+
vol: float
|
| 34 |
+
price: float
|
| 35 |
+
|
| 36 |
+
|
| 37 |
+
SIM_PROFILES: dict[str, SimProfile] = {
|
| 38 |
+
"AAPL": SimProfile(0.24, 0.29, 190.0),
|
| 39 |
+
"MSFT": SimProfile(0.25, 0.27, 410.0),
|
| 40 |
+
"NVDA": SimProfile(0.55, 0.52, 120.0),
|
| 41 |
+
"TSLA": SimProfile(0.30, 0.58, 250.0),
|
| 42 |
+
"AMZN": SimProfile(0.22, 0.33, 180.0),
|
| 43 |
+
"GOOGL": SimProfile(0.20, 0.31, 170.0),
|
| 44 |
+
"META": SimProfile(0.26, 0.40, 500.0),
|
| 45 |
+
"SPY": SimProfile(0.10, 0.16, 550.0),
|
| 46 |
+
"QQQ": SimProfile(0.14, 0.21, 480.0),
|
| 47 |
+
"BTC-USD": SimProfile(0.45, 0.65, 65000.0),
|
| 48 |
+
"ETH-USD": SimProfile(0.35, 0.75, 3000.0),
|
| 49 |
+
"GLD": SimProfile(0.07, 0.14, 200.0),
|
| 50 |
+
"TLT": SimProfile(0.01, 0.15, 95.0),
|
| 51 |
+
}
|
| 52 |
+
|
| 53 |
+
DEFAULT_UNIVERSE = ["SPY", "AAPL", "NVDA", "MSFT", "TSLA", "QQQ", "BTC-USD", "GLD"]
|
| 54 |
+
|
| 55 |
+
|
| 56 |
+
def _seed_for(symbol: str) -> int:
|
| 57 |
+
"""Stable per-symbol seed so a given ticker always simulates identically."""
|
| 58 |
+
digest = hashlib.sha256(symbol.upper().encode()).digest()
|
| 59 |
+
return int.from_bytes(digest[:4], "big")
|
| 60 |
+
|
| 61 |
+
|
| 62 |
+
def _normalise(df: pd.DataFrame) -> pd.DataFrame:
|
| 63 |
+
"""Coerce any loader's output into a clean lowercase OHLCV frame."""
|
| 64 |
+
if isinstance(df.columns, pd.MultiIndex):
|
| 65 |
+
df = df.copy()
|
| 66 |
+
df.columns = [str(c[0]) for c in df.columns]
|
| 67 |
+
df = df.rename(columns={c: str(c).strip().lower().replace(" ", "_") for c in df.columns})
|
| 68 |
+
if "adj_close" in df.columns and "close" not in df.columns:
|
| 69 |
+
df = df.rename(columns={"adj_close": "close"})
|
| 70 |
+
missing = [c for c in OHLCV_COLUMNS if c not in df.columns]
|
| 71 |
+
for col in missing:
|
| 72 |
+
if col == "volume":
|
| 73 |
+
df["volume"] = 0.0
|
| 74 |
+
elif "close" in df.columns:
|
| 75 |
+
df[col] = df["close"]
|
| 76 |
+
else:
|
| 77 |
+
raise ValueError(f"Price data is missing required column: {col}")
|
| 78 |
+
df = df.loc[:, list(OHLCV_COLUMNS)].astype(float)
|
| 79 |
+
if not isinstance(df.index, pd.DatetimeIndex):
|
| 80 |
+
df.index = pd.to_datetime(df.index)
|
| 81 |
+
df.index = df.index.tz_localize(None) if df.index.tz is not None else df.index
|
| 82 |
+
df = df[~df.index.duplicated(keep="last")].sort_index()
|
| 83 |
+
df = df[df["close"] > 0].dropna(subset=["close"])
|
| 84 |
+
return df
|
| 85 |
+
|
| 86 |
+
|
| 87 |
+
def _cache_path(symbol: str, interval: str) -> Path:
|
| 88 |
+
safe = symbol.upper().replace("/", "_")
|
| 89 |
+
return CACHE_DIR / f"{safe}_{interval}.csv"
|
| 90 |
+
|
| 91 |
+
|
| 92 |
+
def _read_cache(symbol: str, interval: str) -> Optional[pd.DataFrame]:
|
| 93 |
+
path = _cache_path(symbol, interval)
|
| 94 |
+
if not path.exists():
|
| 95 |
+
return None
|
| 96 |
+
try:
|
| 97 |
+
return _normalise(pd.read_csv(path, index_col=0, parse_dates=True))
|
| 98 |
+
except Exception as exc: # pragma: no cover - corrupted cache is not worth failing over
|
| 99 |
+
logger.warning("Ignoring unreadable cache %s: %s", path, exc)
|
| 100 |
+
return None
|
| 101 |
+
|
| 102 |
+
|
| 103 |
+
def _write_cache(symbol: str, interval: str, df: pd.DataFrame) -> None:
|
| 104 |
+
try:
|
| 105 |
+
CACHE_DIR.mkdir(parents=True, exist_ok=True)
|
| 106 |
+
df.to_csv(_cache_path(symbol, interval))
|
| 107 |
+
except Exception as exc: # pragma: no cover - a read-only FS must not break the app
|
| 108 |
+
logger.warning("Could not write cache for %s: %s", symbol, exc)
|
| 109 |
+
|
| 110 |
+
|
| 111 |
+
def _download(symbol: str, start: str, end: str | None, interval: str) -> Optional[pd.DataFrame]:
|
| 112 |
+
if not NETWORK_ENABLED:
|
| 113 |
+
return None
|
| 114 |
+
try:
|
| 115 |
+
import yfinance as yf
|
| 116 |
+
except ImportError:
|
| 117 |
+
logger.info("yfinance not installed; using offline data")
|
| 118 |
+
return None
|
| 119 |
+
try:
|
| 120 |
+
raw = yf.download(
|
| 121 |
+
symbol,
|
| 122 |
+
start=start,
|
| 123 |
+
end=end,
|
| 124 |
+
interval=interval,
|
| 125 |
+
progress=False,
|
| 126 |
+
auto_adjust=True,
|
| 127 |
+
threads=False,
|
| 128 |
+
)
|
| 129 |
+
except Exception as exc:
|
| 130 |
+
logger.warning("Download failed for %s: %s", symbol, exc)
|
| 131 |
+
return None
|
| 132 |
+
if raw is None or len(raw) == 0:
|
| 133 |
+
logger.warning("Download for %s returned no rows", symbol)
|
| 134 |
+
return None
|
| 135 |
+
try:
|
| 136 |
+
return _normalise(raw)
|
| 137 |
+
except Exception as exc:
|
| 138 |
+
logger.warning("Could not normalise download for %s: %s", symbol, exc)
|
| 139 |
+
return None
|
| 140 |
+
|
| 141 |
+
|
| 142 |
+
def simulate_ohlcv(
|
| 143 |
+
symbol: str = "SIM",
|
| 144 |
+
start: str = "2015-01-01",
|
| 145 |
+
end: str | None = None,
|
| 146 |
+
interval: str = "1d",
|
| 147 |
+
seed: Optional[int] = None,
|
| 148 |
+
) -> pd.DataFrame:
|
| 149 |
+
"""Generate a deterministic but realistic-looking OHLCV series.
|
| 150 |
+
|
| 151 |
+
This is not geometric Brownian motion with a straight face: it uses a
|
| 152 |
+
two-state (calm / stressed) regime switch, Student-t innovations and
|
| 153 |
+
GARCH-ish vol persistence, so the resulting series has fat tails and
|
| 154 |
+
volatility clustering. That matters, because a strategy tested against
|
| 155 |
+
naive GBM looks far better than it deserves to.
|
| 156 |
+
"""
|
| 157 |
+
profile = SIM_PROFILES.get(symbol.upper(), SimProfile(0.08, 0.25, 100.0))
|
| 158 |
+
rng = np.random.default_rng(_seed_for(symbol) if seed is None else seed)
|
| 159 |
+
|
| 160 |
+
freq = {"1d": "B", "1wk": "W-FRI", "1h": "h"}.get(interval, "B")
|
| 161 |
+
index = pd.date_range(start=start, end=end or pd.Timestamp.today().normalize(), freq=freq)
|
| 162 |
+
n = len(index)
|
| 163 |
+
if n < 50:
|
| 164 |
+
raise ValueError("Simulated range is too short to backtest")
|
| 165 |
+
|
| 166 |
+
ppy = 252 if freq in ("B", "h") else 52
|
| 167 |
+
mu = profile.drift / ppy
|
| 168 |
+
base_vol = profile.vol / np.sqrt(ppy)
|
| 169 |
+
|
| 170 |
+
# Regime chain: calm state is sticky, stressed state is short and violent.
|
| 171 |
+
p_calm_to_stress, p_stress_to_calm = 0.01, 0.06
|
| 172 |
+
regime = np.zeros(n, dtype=int)
|
| 173 |
+
for i in range(1, n):
|
| 174 |
+
flip = rng.random()
|
| 175 |
+
if regime[i - 1] == 0:
|
| 176 |
+
regime[i] = 1 if flip < p_calm_to_stress else 0
|
| 177 |
+
else:
|
| 178 |
+
regime[i] = 0 if flip < p_stress_to_calm else 1
|
| 179 |
+
|
| 180 |
+
# Persistent vol around a regime-dependent level.
|
| 181 |
+
vol = np.empty(n)
|
| 182 |
+
level = np.where(regime == 1, base_vol * 2.4, base_vol * 0.9)
|
| 183 |
+
vol[0] = level[0]
|
| 184 |
+
for i in range(1, n):
|
| 185 |
+
vol[i] = 0.92 * vol[i - 1] + 0.08 * level[i]
|
| 186 |
+
|
| 187 |
+
shocks = rng.standard_t(df=4, size=n) / np.sqrt(2.0) # unit-ish variance, fat tails
|
| 188 |
+
drift = np.where(regime == 1, mu - 3.0 * base_vol**2, mu)
|
| 189 |
+
log_ret = drift + vol * shocks
|
| 190 |
+
close = profile.price * np.exp(np.cumsum(log_ret))
|
| 191 |
+
close = close * (profile.price / close[-1]) # end near the quoted level
|
| 192 |
+
|
| 193 |
+
intrabar = vol * rng.uniform(0.3, 1.1, size=n)
|
| 194 |
+
open_ = close * np.exp(-log_ret * rng.uniform(0.2, 0.8, size=n))
|
| 195 |
+
high = np.maximum(open_, close) * np.exp(np.abs(intrabar))
|
| 196 |
+
low = np.minimum(open_, close) * np.exp(-np.abs(intrabar))
|
| 197 |
+
volume = rng.lognormal(mean=15.5, sigma=0.45, size=n) * (1.0 + 3.0 * regime)
|
| 198 |
+
|
| 199 |
+
return _normalise(
|
| 200 |
+
pd.DataFrame(
|
| 201 |
+
{"open": open_, "high": high, "low": low, "close": close, "volume": volume},
|
| 202 |
+
index=index,
|
| 203 |
+
)
|
| 204 |
+
)
|
| 205 |
+
|
| 206 |
+
|
| 207 |
+
def load_ohlcv(
|
| 208 |
+
symbol: str = "SPY",
|
| 209 |
+
start: str = "2015-01-01",
|
| 210 |
+
end: str | None = None,
|
| 211 |
+
interval: str = "1d",
|
| 212 |
+
source: str = "yahoo",
|
| 213 |
+
) -> MarketData:
|
| 214 |
+
"""Load OHLCV for ``symbol``.
|
| 215 |
+
|
| 216 |
+
Default ``source='yahoo'`` requires a Yahoo download. ``auto`` still falls
|
| 217 |
+
back to cache then the simulator (Hugging Face Space). ``synthetic`` is tests only.
|
| 218 |
+
"""
|
| 219 |
+
symbol = (symbol or "SPY").strip().upper()
|
| 220 |
+
|
| 221 |
+
if source == "synthetic":
|
| 222 |
+
df = simulate_ohlcv(symbol, start, end, interval)
|
| 223 |
+
return MarketData(symbol, df, "synthetic", interval, "Simulated prices (requested).")
|
| 224 |
+
|
| 225 |
+
if source in ("yahoo", "live", "auto"):
|
| 226 |
+
df = _download(symbol, start, end, interval)
|
| 227 |
+
if df is not None and len(df) > 50:
|
| 228 |
+
_write_cache(symbol, interval, df)
|
| 229 |
+
return MarketData(symbol, df, "yfinance", interval, "Live data from Yahoo Finance.")
|
| 230 |
+
if source in ("yahoo", "live"):
|
| 231 |
+
raise RuntimeError(
|
| 232 |
+
f"Yahoo returned no usable bars for {symbol}. "
|
| 233 |
+
"Check the ticker, date range, and network. "
|
| 234 |
+
"Pass source='synthetic' only for offline tests."
|
| 235 |
+
)
|
| 236 |
+
|
| 237 |
+
cached = _read_cache(symbol, interval)
|
| 238 |
+
if cached is not None and len(cached) > 50:
|
| 239 |
+
window = cached.loc[str(start) : str(end)] if end else cached.loc[str(start) :]
|
| 240 |
+
if len(window) > 50:
|
| 241 |
+
return MarketData(symbol, window, "bundled", interval, "Cached data (network unavailable).")
|
| 242 |
+
|
| 243 |
+
df = simulate_ohlcv(symbol, start, end, interval)
|
| 244 |
+
return MarketData(
|
| 245 |
+
symbol,
|
| 246 |
+
df,
|
| 247 |
+
"synthetic",
|
| 248 |
+
interval,
|
| 249 |
+
f"Live data for {symbol} was unavailable, so this run uses a deterministic "
|
| 250 |
+
"market simulator with fat tails and volatility clustering. The statistics "
|
| 251 |
+
"below are still valid — they are just measured on a simulated market.",
|
| 252 |
+
)
|
algotrader/engine.py
ADDED
|
@@ -0,0 +1,140 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Vectorised, look-ahead-free backtest engine.
|
| 2 |
+
|
| 3 |
+
Contract
|
| 4 |
+
--------
|
| 5 |
+
A strategy emits ``target[t]``: the exposure it wants, decided using only
|
| 6 |
+
information available at the close of bar ``t``. The engine holds
|
| 7 |
+
``position[t] = target[t - lag]`` during bar ``t`` and credits it with that
|
| 8 |
+
bar's close-to-close return. With the default ``lag=1`` this means "decide on
|
| 9 |
+
today's close, hold the position through tomorrow" -- the single place where
|
| 10 |
+
look-ahead could sneak in, and it is one line.
|
| 11 |
+
|
| 12 |
+
Costs are charged on exposure *changes*, so a strategy that flips daily pays
|
| 13 |
+
for it. Short exposure additionally accrues a borrow fee.
|
| 14 |
+
"""
|
| 15 |
+
|
| 16 |
+
from __future__ import annotations
|
| 17 |
+
|
| 18 |
+
from typing import Optional
|
| 19 |
+
|
| 20 |
+
import numpy as np
|
| 21 |
+
import pandas as pd
|
| 22 |
+
|
| 23 |
+
from .metrics import compute_metrics, infer_periods_per_year
|
| 24 |
+
from .types import BacktestResult, CostModel
|
| 25 |
+
|
| 26 |
+
__all__ = ["run_backtest", "bars_to_returns"]
|
| 27 |
+
|
| 28 |
+
|
| 29 |
+
def bars_to_returns(df: pd.DataFrame) -> pd.Series:
|
| 30 |
+
"""Close-to-close simple returns."""
|
| 31 |
+
return df["close"].astype(float).pct_change().fillna(0.0)
|
| 32 |
+
|
| 33 |
+
|
| 34 |
+
def run_backtest(
|
| 35 |
+
df: pd.DataFrame,
|
| 36 |
+
target: pd.Series,
|
| 37 |
+
costs: CostModel | None = None,
|
| 38 |
+
lag: int = 1,
|
| 39 |
+
max_leverage: float = 1.0,
|
| 40 |
+
allow_short: bool = True,
|
| 41 |
+
initial_capital: float = 100_000.0,
|
| 42 |
+
periods_per_year: Optional[int] = None,
|
| 43 |
+
rf: float = 0.0,
|
| 44 |
+
meta: Optional[dict] = None,
|
| 45 |
+
) -> BacktestResult:
|
| 46 |
+
"""Run one backtest and return equity, returns and the full metric bundle."""
|
| 47 |
+
if df.empty:
|
| 48 |
+
raise ValueError("Cannot backtest an empty price frame")
|
| 49 |
+
if lag < 1:
|
| 50 |
+
raise ValueError("lag must be >= 1; lag=0 would trade on unavailable information")
|
| 51 |
+
|
| 52 |
+
costs = costs or CostModel()
|
| 53 |
+
ppy = periods_per_year or infer_periods_per_year(df.index)
|
| 54 |
+
|
| 55 |
+
asset_ret = bars_to_returns(df)
|
| 56 |
+
|
| 57 |
+
target = target.reindex(df.index).astype(float).fillna(0.0)
|
| 58 |
+
lower = -max_leverage if allow_short else 0.0
|
| 59 |
+
target = target.clip(lower, max_leverage)
|
| 60 |
+
|
| 61 |
+
position = target.shift(lag).fillna(0.0)
|
| 62 |
+
|
| 63 |
+
gross = position * asset_ret
|
| 64 |
+
|
| 65 |
+
# Turnover is measured against the *drifted* weight, not the previous
|
| 66 |
+
# target. Holding a full-notional long needs no rebalancing (the position
|
| 67 |
+
# and the portfolio grow together), but a short does: lose 10% on a 100%
|
| 68 |
+
# short and the weight drifts to -82%, so staying at -100% costs a trade.
|
| 69 |
+
# See portfolio.py for the same formula in matrix form.
|
| 70 |
+
growth = (1.0 + gross).replace(0.0, np.nan)
|
| 71 |
+
drifted = (position * (1.0 + asset_ret)) / growth
|
| 72 |
+
previous = drifted.shift(1).fillna(0.0)
|
| 73 |
+
traded = position - previous
|
| 74 |
+
trade_cost = traded.abs() * (costs.one_way_bps / 1e4)
|
| 75 |
+
|
| 76 |
+
borrow_cost = position.clip(upper=0.0).abs() * (costs.short_borrow_bps / 1e4) / ppy
|
| 77 |
+
total_cost = trade_cost + borrow_cost
|
| 78 |
+
|
| 79 |
+
net = gross - total_cost
|
| 80 |
+
equity = initial_capital * (1.0 + net).cumprod()
|
| 81 |
+
benchmark_equity = initial_capital * (1.0 + asset_ret).cumprod()
|
| 82 |
+
|
| 83 |
+
result = BacktestResult(
|
| 84 |
+
equity=equity,
|
| 85 |
+
returns=net,
|
| 86 |
+
gross_returns=gross,
|
| 87 |
+
position=position,
|
| 88 |
+
target=target,
|
| 89 |
+
costs=total_cost,
|
| 90 |
+
benchmark_equity=benchmark_equity,
|
| 91 |
+
metrics=compute_metrics(net, equity, position, ppy, rf),
|
| 92 |
+
benchmark_metrics=compute_metrics(asset_ret, benchmark_equity, None, ppy, rf),
|
| 93 |
+
meta={
|
| 94 |
+
"lag": lag,
|
| 95 |
+
"commission_bps": costs.commission_bps,
|
| 96 |
+
"slippage_bps": costs.slippage_bps,
|
| 97 |
+
"short_borrow_bps": costs.short_borrow_bps,
|
| 98 |
+
"max_leverage": max_leverage,
|
| 99 |
+
"allow_short": allow_short,
|
| 100 |
+
"initial_capital": initial_capital,
|
| 101 |
+
"periods_per_year": ppy,
|
| 102 |
+
**(meta or {}),
|
| 103 |
+
},
|
| 104 |
+
)
|
| 105 |
+
result.metrics["cost_drag_ann"] = float(total_cost.sum() / max(result.metrics.get("years", 1e-9), 1e-9))
|
| 106 |
+
result.metrics["gross_sharpe"] = float(
|
| 107 |
+
compute_metrics(gross, initial_capital * (1.0 + gross).cumprod(), None, ppy, rf).get("sharpe", 0.0)
|
| 108 |
+
)
|
| 109 |
+
return result
|
| 110 |
+
|
| 111 |
+
|
| 112 |
+
def fast_sharpe(
|
| 113 |
+
asset_ret: np.ndarray,
|
| 114 |
+
target: np.ndarray,
|
| 115 |
+
one_way_bps: float,
|
| 116 |
+
lag: int,
|
| 117 |
+
periods_per_year: int,
|
| 118 |
+
) -> float:
|
| 119 |
+
"""Numpy-only Sharpe for hot loops (permutation tests, PBO grids).
|
| 120 |
+
|
| 121 |
+
Mirrors :func:`run_backtest` exactly for the no-borrow case; it exists only
|
| 122 |
+
because building a DataFrame 1000 times is the difference between a Space
|
| 123 |
+
that answers in 4 seconds and one nobody waits for.
|
| 124 |
+
"""
|
| 125 |
+
n = asset_ret.size
|
| 126 |
+
position = np.empty(n, dtype=float)
|
| 127 |
+
position[:lag] = 0.0
|
| 128 |
+
position[lag:] = target[:-lag] if lag else target
|
| 129 |
+
gross = position * asset_ret
|
| 130 |
+
traded = np.empty(n, dtype=float)
|
| 131 |
+
traded[0] = position[0]
|
| 132 |
+
traded[1:] = np.diff(position)
|
| 133 |
+
net = gross - np.abs(traded) * (one_way_bps / 1e4)
|
| 134 |
+
net = net[np.isfinite(net)]
|
| 135 |
+
if net.size < 2:
|
| 136 |
+
return 0.0
|
| 137 |
+
sd = net.std(ddof=1)
|
| 138 |
+
if not np.isfinite(sd) or sd < 1e-12:
|
| 139 |
+
return 0.0
|
| 140 |
+
return float(net.mean() / sd * np.sqrt(periods_per_year))
|
algotrader/indicators.py
ADDED
|
@@ -0,0 +1,102 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Vectorised technical indicators.
|
| 2 |
+
|
| 3 |
+
Every function takes and returns pandas objects aligned to the input index, and
|
| 4 |
+
every one of them is causal: the value at bar ``t`` uses only data up to and
|
| 5 |
+
including ``t``. That property is what makes the backtest engine's single
|
| 6 |
+
``shift`` enough to guarantee no look-ahead.
|
| 7 |
+
"""
|
| 8 |
+
|
| 9 |
+
from __future__ import annotations
|
| 10 |
+
|
| 11 |
+
import numpy as np
|
| 12 |
+
import pandas as pd
|
| 13 |
+
|
| 14 |
+
__all__ = [
|
| 15 |
+
"sma",
|
| 16 |
+
"ema",
|
| 17 |
+
"rsi",
|
| 18 |
+
"macd",
|
| 19 |
+
"bollinger",
|
| 20 |
+
"atr",
|
| 21 |
+
"donchian",
|
| 22 |
+
"zscore",
|
| 23 |
+
"roc",
|
| 24 |
+
"realised_vol",
|
| 25 |
+
]
|
| 26 |
+
|
| 27 |
+
|
| 28 |
+
def sma(series: pd.Series, window: int) -> pd.Series:
|
| 29 |
+
return series.rolling(window, min_periods=window).mean()
|
| 30 |
+
|
| 31 |
+
|
| 32 |
+
def ema(series: pd.Series, window: int) -> pd.Series:
|
| 33 |
+
return series.ewm(span=window, adjust=False, min_periods=window).mean()
|
| 34 |
+
|
| 35 |
+
|
| 36 |
+
def rsi(series: pd.Series, window: int = 14) -> pd.Series:
|
| 37 |
+
"""Wilder's RSI."""
|
| 38 |
+
delta = series.diff()
|
| 39 |
+
gain = delta.clip(lower=0.0)
|
| 40 |
+
loss = -delta.clip(upper=0.0)
|
| 41 |
+
avg_gain = gain.ewm(alpha=1.0 / window, adjust=False, min_periods=window).mean()
|
| 42 |
+
avg_loss = loss.ewm(alpha=1.0 / window, adjust=False, min_periods=window).mean()
|
| 43 |
+
rs = avg_gain / avg_loss.replace(0.0, np.nan)
|
| 44 |
+
out = 100.0 - (100.0 / (1.0 + rs))
|
| 45 |
+
# avg_loss == 0 leaves rs undefined: an all-gain window is RSI 100, and a
|
| 46 |
+
# perfectly flat window (no gains either) is RSI 50.
|
| 47 |
+
flat = (avg_gain == 0.0) & (avg_loss == 0.0)
|
| 48 |
+
out = out.mask((avg_loss == 0.0) & (avg_gain > 0.0), 100.0)
|
| 49 |
+
out = out.mask(flat, 50.0)
|
| 50 |
+
return out.where(avg_gain.notna())
|
| 51 |
+
|
| 52 |
+
|
| 53 |
+
def macd(
|
| 54 |
+
series: pd.Series, fast: int = 12, slow: int = 26, signal: int = 9
|
| 55 |
+
) -> tuple[pd.Series, pd.Series, pd.Series]:
|
| 56 |
+
"""Returns ``(macd_line, signal_line, histogram)``."""
|
| 57 |
+
macd_line = ema(series, fast) - ema(series, slow)
|
| 58 |
+
signal_line = macd_line.ewm(span=signal, adjust=False, min_periods=signal).mean()
|
| 59 |
+
return macd_line, signal_line, macd_line - signal_line
|
| 60 |
+
|
| 61 |
+
|
| 62 |
+
def bollinger(
|
| 63 |
+
series: pd.Series, window: int = 20, k: float = 2.0
|
| 64 |
+
) -> tuple[pd.Series, pd.Series, pd.Series]:
|
| 65 |
+
"""Returns ``(lower, middle, upper)``."""
|
| 66 |
+
mid = sma(series, window)
|
| 67 |
+
sd = series.rolling(window, min_periods=window).std(ddof=0)
|
| 68 |
+
return mid - k * sd, mid, mid + k * sd
|
| 69 |
+
|
| 70 |
+
|
| 71 |
+
def atr(df: pd.DataFrame, window: int = 14) -> pd.Series:
|
| 72 |
+
prev_close = df["close"].shift(1)
|
| 73 |
+
tr = pd.concat(
|
| 74 |
+
[
|
| 75 |
+
df["high"] - df["low"],
|
| 76 |
+
(df["high"] - prev_close).abs(),
|
| 77 |
+
(df["low"] - prev_close).abs(),
|
| 78 |
+
],
|
| 79 |
+
axis=1,
|
| 80 |
+
).max(axis=1)
|
| 81 |
+
return tr.ewm(alpha=1.0 / window, adjust=False, min_periods=window).mean()
|
| 82 |
+
|
| 83 |
+
|
| 84 |
+
def donchian(df: pd.DataFrame, window: int = 20) -> tuple[pd.Series, pd.Series]:
|
| 85 |
+
"""Rolling channel excluding the current bar, so a breakout test is causal."""
|
| 86 |
+
upper = df["high"].rolling(window, min_periods=window).max().shift(1)
|
| 87 |
+
lower = df["low"].rolling(window, min_periods=window).min().shift(1)
|
| 88 |
+
return lower, upper
|
| 89 |
+
|
| 90 |
+
|
| 91 |
+
def zscore(series: pd.Series, window: int = 20) -> pd.Series:
|
| 92 |
+
mean = series.rolling(window, min_periods=window).mean()
|
| 93 |
+
sd = series.rolling(window, min_periods=window).std(ddof=0)
|
| 94 |
+
return (series - mean) / sd.replace(0.0, np.nan)
|
| 95 |
+
|
| 96 |
+
|
| 97 |
+
def roc(series: pd.Series, window: int = 20) -> pd.Series:
|
| 98 |
+
return series.pct_change(window)
|
| 99 |
+
|
| 100 |
+
|
| 101 |
+
def realised_vol(returns: pd.Series, window: int = 20, periods_per_year: int = 252) -> pd.Series:
|
| 102 |
+
return returns.rolling(window, min_periods=window).std(ddof=0) * np.sqrt(periods_per_year)
|
algotrader/lab.py
ADDED
|
@@ -0,0 +1,328 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""The Lab: one call that runs a backtest and then tries to disprove it.
|
| 2 |
+
|
| 3 |
+
This is the module both the Gradio Space and the CLI drive. Keeping the whole
|
| 4 |
+
pipeline here means the app and the command line can never disagree about what
|
| 5 |
+
a Reality Score means.
|
| 6 |
+
"""
|
| 7 |
+
|
| 8 |
+
from __future__ import annotations
|
| 9 |
+
|
| 10 |
+
import logging
|
| 11 |
+
from dataclasses import dataclass, field
|
| 12 |
+
from typing import Callable, Dict, List, Optional
|
| 13 |
+
|
| 14 |
+
import numpy as np
|
| 15 |
+
import pandas as pd
|
| 16 |
+
|
| 17 |
+
from .data import load_ohlcv
|
| 18 |
+
from .engine import run_backtest
|
| 19 |
+
from .metrics import infer_periods_per_year
|
| 20 |
+
from .strategies import Strategy, get_strategy, list_strategies
|
| 21 |
+
from .types import BacktestResult, CostModel, MarketData
|
| 22 |
+
from .validation.deflated_sharpe import deflated_sharpe_ratio, min_track_record_length
|
| 23 |
+
from .validation.pbo import probability_of_backtest_overfitting
|
| 24 |
+
from .validation.permutation import PermutationResult, permutation_test
|
| 25 |
+
from .validation.walkforward import walk_forward
|
| 26 |
+
from .verdict import reality_score
|
| 27 |
+
|
| 28 |
+
logger = logging.getLogger(__name__)
|
| 29 |
+
|
| 30 |
+
__all__ = ["LabConfig", "LabReport", "run_lab", "run_arena"]
|
| 31 |
+
|
| 32 |
+
ProgressFn = Optional[Callable[[float, str], None]]
|
| 33 |
+
|
| 34 |
+
|
| 35 |
+
@dataclass
|
| 36 |
+
class LabConfig:
|
| 37 |
+
symbol: str = "SPY"
|
| 38 |
+
start: str = "2015-01-01"
|
| 39 |
+
end: Optional[str] = None
|
| 40 |
+
interval: str = "1d"
|
| 41 |
+
source: str = "yahoo"
|
| 42 |
+
|
| 43 |
+
strategy: str = "sma_cross"
|
| 44 |
+
params: Dict[str, float] = field(default_factory=dict)
|
| 45 |
+
|
| 46 |
+
commission_bps: float = 1.0
|
| 47 |
+
slippage_bps: float = 2.0
|
| 48 |
+
short_borrow_bps: float = 50.0
|
| 49 |
+
lag: int = 1
|
| 50 |
+
allow_short: bool = True
|
| 51 |
+
max_leverage: float = 1.0
|
| 52 |
+
capital: float = 100_000.0
|
| 53 |
+
|
| 54 |
+
n_permutations: int = 250
|
| 55 |
+
permutation_method: str = "permute"
|
| 56 |
+
block_size: int = 20
|
| 57 |
+
wf_folds: int = 5
|
| 58 |
+
pbo_splits: int = 8
|
| 59 |
+
grid_limit: int = 40
|
| 60 |
+
seed: int = 0
|
| 61 |
+
|
| 62 |
+
def costs(self, multiplier: float = 1.0) -> CostModel:
|
| 63 |
+
return CostModel(
|
| 64 |
+
commission_bps=self.commission_bps * multiplier,
|
| 65 |
+
slippage_bps=self.slippage_bps * multiplier,
|
| 66 |
+
short_borrow_bps=self.short_borrow_bps * multiplier,
|
| 67 |
+
)
|
| 68 |
+
|
| 69 |
+
|
| 70 |
+
@dataclass
|
| 71 |
+
class LabReport:
|
| 72 |
+
config: LabConfig
|
| 73 |
+
market: MarketData
|
| 74 |
+
strategy: Strategy
|
| 75 |
+
params: Dict[str, float]
|
| 76 |
+
backtest: BacktestResult
|
| 77 |
+
permutation: Optional[PermutationResult] = None
|
| 78 |
+
dsr: Dict[str, float] = field(default_factory=dict)
|
| 79 |
+
pbo: Dict[str, object] = field(default_factory=dict)
|
| 80 |
+
walkforward: Dict[str, object] = field(default_factory=dict)
|
| 81 |
+
trials: Dict[str, object] = field(default_factory=dict)
|
| 82 |
+
verdict: Dict[str, object] = field(default_factory=dict)
|
| 83 |
+
cost_stress: Dict[str, float] = field(default_factory=dict)
|
| 84 |
+
benchmark_correlation: float = float("nan")
|
| 85 |
+
|
| 86 |
+
|
| 87 |
+
def _trial_matrix(
|
| 88 |
+
df: pd.DataFrame,
|
| 89 |
+
strategy: Strategy,
|
| 90 |
+
cfg: LabConfig,
|
| 91 |
+
progress: ProgressFn = None,
|
| 92 |
+
) -> tuple[np.ndarray, List[float], List[str]]:
|
| 93 |
+
"""Backtest every parameter combination a researcher would plausibly try.
|
| 94 |
+
|
| 95 |
+
The resulting ``T x N`` return matrix feeds both the Deflated Sharpe (how
|
| 96 |
+
many variants were tried, and how spread out were they) and PBO.
|
| 97 |
+
"""
|
| 98 |
+
grid = strategy.grid(limit=cfg.grid_limit)
|
| 99 |
+
costs = cfg.costs()
|
| 100 |
+
columns, sharpes, labels = [], [], []
|
| 101 |
+
|
| 102 |
+
for i, params in enumerate(grid):
|
| 103 |
+
target = strategy.generate(df, params)
|
| 104 |
+
result = run_backtest(
|
| 105 |
+
df, target, costs=costs, lag=cfg.lag,
|
| 106 |
+
max_leverage=cfg.max_leverage, allow_short=cfg.allow_short,
|
| 107 |
+
)
|
| 108 |
+
columns.append(result.returns.to_numpy(dtype=float))
|
| 109 |
+
sharpes.append(result.sharpe)
|
| 110 |
+
labels.append(", ".join(f"{k}={v}" for k, v in params.items()) or "default")
|
| 111 |
+
if progress is not None and i % 5 == 0:
|
| 112 |
+
progress((i + 1) / max(len(grid), 1), f"Variant {i + 1}/{len(grid)}")
|
| 113 |
+
|
| 114 |
+
matrix = np.column_stack(columns) if columns else np.zeros((len(df), 0))
|
| 115 |
+
return matrix, sharpes, labels
|
| 116 |
+
|
| 117 |
+
|
| 118 |
+
def run_lab(cfg: LabConfig, progress: ProgressFn = None) -> LabReport:
|
| 119 |
+
"""Run the full honesty pipeline for one strategy on one symbol."""
|
| 120 |
+
|
| 121 |
+
def step(fraction: float, message: str) -> None:
|
| 122 |
+
if progress is not None:
|
| 123 |
+
progress(min(max(fraction, 0.0), 1.0), message)
|
| 124 |
+
|
| 125 |
+
step(0.02, "Loading market data")
|
| 126 |
+
market = load_ohlcv(cfg.symbol, cfg.start, cfg.end, cfg.interval, cfg.source)
|
| 127 |
+
df = market.df
|
| 128 |
+
if len(df) < 120:
|
| 129 |
+
raise ValueError(
|
| 130 |
+
f"Only {len(df)} bars available for {cfg.symbol}. "
|
| 131 |
+
"Widen the date range — anything shorter cannot be validated."
|
| 132 |
+
)
|
| 133 |
+
|
| 134 |
+
strategy = get_strategy(cfg.strategy)
|
| 135 |
+
params = strategy.clean(cfg.params)
|
| 136 |
+
ppy = infer_periods_per_year(df.index)
|
| 137 |
+
|
| 138 |
+
step(0.10, "Running the backtest")
|
| 139 |
+
target = strategy.generate(df, params)
|
| 140 |
+
backtest = run_backtest(
|
| 141 |
+
df,
|
| 142 |
+
target,
|
| 143 |
+
costs=cfg.costs(),
|
| 144 |
+
lag=cfg.lag,
|
| 145 |
+
max_leverage=cfg.max_leverage,
|
| 146 |
+
allow_short=cfg.allow_short,
|
| 147 |
+
initial_capital=cfg.capital,
|
| 148 |
+
periods_per_year=ppy,
|
| 149 |
+
meta={"symbol": market.symbol, "strategy": strategy.key, "params": params},
|
| 150 |
+
)
|
| 151 |
+
|
| 152 |
+
step(0.16, "Stress-testing costs")
|
| 153 |
+
stressed = run_backtest(
|
| 154 |
+
df, target, costs=cfg.costs(3.0), lag=cfg.lag,
|
| 155 |
+
max_leverage=cfg.max_leverage, allow_short=cfg.allow_short,
|
| 156 |
+
periods_per_year=ppy,
|
| 157 |
+
)
|
| 158 |
+
base_sharpe = backtest.sharpe
|
| 159 |
+
cost_stress_ratio = float(stressed.sharpe / base_sharpe) if base_sharpe > 1e-9 else 0.0
|
| 160 |
+
cost_stress = {
|
| 161 |
+
"sharpe_1x": base_sharpe,
|
| 162 |
+
"sharpe_3x": stressed.sharpe,
|
| 163 |
+
"ratio": cost_stress_ratio,
|
| 164 |
+
"return_3x": float(stressed.metrics.get("total_return", 0.0)),
|
| 165 |
+
}
|
| 166 |
+
|
| 167 |
+
step(0.22, "Backtesting every parameter variant")
|
| 168 |
+
matrix, trial_sharpes, labels = _trial_matrix(
|
| 169 |
+
df, strategy, cfg, lambda f, m: step(0.22 + 0.18 * f, m)
|
| 170 |
+
)
|
| 171 |
+
n_trials = max(len(trial_sharpes), 1)
|
| 172 |
+
|
| 173 |
+
step(0.42, "Deflating the Sharpe ratio for selection bias")
|
| 174 |
+
dsr = deflated_sharpe_ratio(
|
| 175 |
+
backtest.returns.to_numpy(dtype=float),
|
| 176 |
+
sharpe_annual=base_sharpe,
|
| 177 |
+
periods_per_year=ppy,
|
| 178 |
+
n_trials=n_trials,
|
| 179 |
+
trial_sharpes=trial_sharpes if n_trials > 1 else None,
|
| 180 |
+
)
|
| 181 |
+
mtrl = min_track_record_length(
|
| 182 |
+
dsr["sr_per_period"], dsr["n_obs"], dsr["skew"], dsr["kurtosis"],
|
| 183 |
+
benchmark=dsr["threshold_sr_per_period"],
|
| 184 |
+
)
|
| 185 |
+
dsr["min_track_record_bars"] = mtrl
|
| 186 |
+
dsr["min_track_record_years"] = float(mtrl / ppy) if np.isfinite(mtrl) else float("inf")
|
| 187 |
+
|
| 188 |
+
step(0.46, "Measuring backtest overfitting")
|
| 189 |
+
pbo = probability_of_backtest_overfitting(matrix, n_splits=cfg.pbo_splits, labels=labels)
|
| 190 |
+
|
| 191 |
+
step(0.50, "Shuffling the market")
|
| 192 |
+
permutation = None
|
| 193 |
+
if cfg.n_permutations > 0:
|
| 194 |
+
permutation = permutation_test(
|
| 195 |
+
df,
|
| 196 |
+
lambda frame: strategy.generate(frame, params),
|
| 197 |
+
n_permutations=cfg.n_permutations,
|
| 198 |
+
method=cfg.permutation_method,
|
| 199 |
+
block=cfg.block_size,
|
| 200 |
+
costs=cfg.costs(),
|
| 201 |
+
lag=cfg.lag,
|
| 202 |
+
max_leverage=cfg.max_leverage,
|
| 203 |
+
allow_short=cfg.allow_short,
|
| 204 |
+
seed=cfg.seed,
|
| 205 |
+
observed=base_sharpe,
|
| 206 |
+
progress=lambda f, m: step(0.50 + 0.32 * f, m),
|
| 207 |
+
)
|
| 208 |
+
|
| 209 |
+
step(0.84, "Walking the strategy forward")
|
| 210 |
+
wf = walk_forward(
|
| 211 |
+
df, strategy, n_folds=cfg.wf_folds, costs=cfg.costs(), lag=cfg.lag,
|
| 212 |
+
max_leverage=cfg.max_leverage, allow_short=cfg.allow_short,
|
| 213 |
+
grid_limit=min(cfg.grid_limit, 24),
|
| 214 |
+
progress=lambda f, m: step(0.84 + 0.12 * f, m),
|
| 215 |
+
)
|
| 216 |
+
|
| 217 |
+
bench_corr = float(
|
| 218 |
+
pd.Series(backtest.returns).corr(backtest.benchmark_equity.pct_change().fillna(0.0))
|
| 219 |
+
)
|
| 220 |
+
|
| 221 |
+
step(0.98, "Grading")
|
| 222 |
+
verdict = reality_score(
|
| 223 |
+
metrics=backtest.metrics,
|
| 224 |
+
benchmark_metrics=backtest.benchmark_metrics,
|
| 225 |
+
p_value=permutation.p_value if permutation else None,
|
| 226 |
+
dsr=dsr.get("dsr"),
|
| 227 |
+
pbo=pbo.get("pbo"),
|
| 228 |
+
wf_efficiency=wf.get("efficiency"),
|
| 229 |
+
wf_win_rate=wf.get("oos_win_rate"),
|
| 230 |
+
cost_stress_ratio=cost_stress_ratio,
|
| 231 |
+
benchmark_correlation=bench_corr,
|
| 232 |
+
)
|
| 233 |
+
|
| 234 |
+
step(1.0, "Done")
|
| 235 |
+
return LabReport(
|
| 236 |
+
config=cfg,
|
| 237 |
+
market=market,
|
| 238 |
+
strategy=strategy,
|
| 239 |
+
params=params,
|
| 240 |
+
backtest=backtest,
|
| 241 |
+
permutation=permutation,
|
| 242 |
+
dsr=dsr,
|
| 243 |
+
pbo=pbo,
|
| 244 |
+
walkforward=wf,
|
| 245 |
+
trials={"n": n_trials, "sharpes": trial_sharpes, "labels": labels, "matrix_shape": matrix.shape},
|
| 246 |
+
verdict=verdict,
|
| 247 |
+
cost_stress=cost_stress,
|
| 248 |
+
benchmark_correlation=bench_corr,
|
| 249 |
+
)
|
| 250 |
+
|
| 251 |
+
|
| 252 |
+
def run_arena(
|
| 253 |
+
cfg: LabConfig,
|
| 254 |
+
strategy_keys: Optional[List[str]] = None,
|
| 255 |
+
n_permutations: int = 120,
|
| 256 |
+
progress: ProgressFn = None,
|
| 257 |
+
) -> tuple[pd.DataFrame, MarketData, Dict[str, BacktestResult]]:
|
| 258 |
+
"""Race every strategy on the same market, ranked by evidence not returns.
|
| 259 |
+
|
| 260 |
+
Buy & hold and the coin flip stay in the field on purpose: a leaderboard
|
| 261 |
+
without a control group is marketing, not measurement.
|
| 262 |
+
"""
|
| 263 |
+
market = load_ohlcv(cfg.symbol, cfg.start, cfg.end, cfg.interval, cfg.source)
|
| 264 |
+
df = market.df
|
| 265 |
+
ppy = infer_periods_per_year(df.index)
|
| 266 |
+
costs = cfg.costs()
|
| 267 |
+
|
| 268 |
+
keys = strategy_keys or [s.key for s in list_strategies()]
|
| 269 |
+
rows, curves = [], {}
|
| 270 |
+
|
| 271 |
+
for i, key in enumerate(keys):
|
| 272 |
+
strategy = get_strategy(key)
|
| 273 |
+
params = strategy.defaults()
|
| 274 |
+
target = strategy.generate(df, params)
|
| 275 |
+
result = run_backtest(
|
| 276 |
+
df, target, costs=costs, lag=cfg.lag, max_leverage=cfg.max_leverage,
|
| 277 |
+
allow_short=cfg.allow_short, initial_capital=cfg.capital, periods_per_year=ppy,
|
| 278 |
+
)
|
| 279 |
+
curves[key] = result
|
| 280 |
+
|
| 281 |
+
p_value = None
|
| 282 |
+
if n_permutations > 0:
|
| 283 |
+
p_value = permutation_test(
|
| 284 |
+
df,
|
| 285 |
+
lambda frame, s=strategy, p=params: s.generate(frame, p),
|
| 286 |
+
n_permutations=n_permutations,
|
| 287 |
+
method=cfg.permutation_method,
|
| 288 |
+
block=cfg.block_size,
|
| 289 |
+
costs=costs,
|
| 290 |
+
lag=cfg.lag,
|
| 291 |
+
max_leverage=cfg.max_leverage,
|
| 292 |
+
allow_short=cfg.allow_short,
|
| 293 |
+
seed=cfg.seed,
|
| 294 |
+
observed=result.sharpe,
|
| 295 |
+
).p_value
|
| 296 |
+
|
| 297 |
+
grid_size = len(strategy.grid(limit=cfg.grid_limit))
|
| 298 |
+
dsr = deflated_sharpe_ratio(
|
| 299 |
+
result.returns.to_numpy(dtype=float),
|
| 300 |
+
sharpe_annual=result.sharpe,
|
| 301 |
+
periods_per_year=ppy,
|
| 302 |
+
n_trials=grid_size,
|
| 303 |
+
)
|
| 304 |
+
|
| 305 |
+
rows.append(
|
| 306 |
+
{
|
| 307 |
+
"Strategy": strategy.name,
|
| 308 |
+
"key": key,
|
| 309 |
+
"Family": strategy.family,
|
| 310 |
+
"Return": result.metrics.get("total_return", 0.0),
|
| 311 |
+
"CAGR": result.metrics.get("cagr", 0.0),
|
| 312 |
+
"Sharpe": result.sharpe,
|
| 313 |
+
"MaxDD": result.metrics.get("max_drawdown", 0.0),
|
| 314 |
+
"Trades": int(result.metrics.get("n_trades", 0)),
|
| 315 |
+
"p-value": p_value if p_value is not None else float("nan"),
|
| 316 |
+
"DSR": dsr["dsr"],
|
| 317 |
+
}
|
| 318 |
+
)
|
| 319 |
+
if progress is not None:
|
| 320 |
+
progress((i + 1) / len(keys), f"{strategy.name} ({i + 1}/{len(keys)})")
|
| 321 |
+
|
| 322 |
+
table = pd.DataFrame(rows)
|
| 323 |
+
if not table.empty:
|
| 324 |
+
# Rank by evidence: a high Sharpe with a p-value of 0.4 is not a win.
|
| 325 |
+
table["Evidence"] = (1.0 - table["p-value"].fillna(0.5)) * table["DSR"]
|
| 326 |
+
table = table.sort_values("Evidence", ascending=False).reset_index(drop=True)
|
| 327 |
+
table.insert(0, "#", table.index + 1)
|
| 328 |
+
return table, market, curves
|
algotrader/metrics.py
ADDED
|
@@ -0,0 +1,157 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Performance and risk metrics.
|
| 2 |
+
|
| 3 |
+
All ratios are computed from *net* per-bar returns and annualised with the
|
| 4 |
+
periodicity inferred from the index, so daily / hourly / minute series all get
|
| 5 |
+
comparable numbers.
|
| 6 |
+
"""
|
| 7 |
+
|
| 8 |
+
from __future__ import annotations
|
| 9 |
+
|
| 10 |
+
from typing import Dict
|
| 11 |
+
|
| 12 |
+
import numpy as np
|
| 13 |
+
import pandas as pd
|
| 14 |
+
|
| 15 |
+
__all__ = [
|
| 16 |
+
"infer_periods_per_year",
|
| 17 |
+
"sharpe_ratio",
|
| 18 |
+
"sortino_ratio",
|
| 19 |
+
"max_drawdown",
|
| 20 |
+
"drawdown_series",
|
| 21 |
+
"compute_metrics",
|
| 22 |
+
]
|
| 23 |
+
|
| 24 |
+
_SECONDS_PER_YEAR = 365.25 * 24 * 3600
|
| 25 |
+
_TRADING_DAYS = 252
|
| 26 |
+
|
| 27 |
+
# A return stream with dispersion below this is constant to floating-point
|
| 28 |
+
# noise. Without an absolute floor, a flat series divides by ~1e-19 and reports
|
| 29 |
+
# a Sharpe of 1e16 -- the exact kind of nonsense number this project exists to
|
| 30 |
+
# catch, so it must not originate here.
|
| 31 |
+
_DEGENERATE_SD = 1e-12
|
| 32 |
+
|
| 33 |
+
|
| 34 |
+
def infer_periods_per_year(index: pd.Index) -> int:
|
| 35 |
+
"""Guess bars-per-year from an index, defaulting to daily trading bars."""
|
| 36 |
+
if not isinstance(index, pd.DatetimeIndex) or len(index) < 3:
|
| 37 |
+
return _TRADING_DAYS
|
| 38 |
+
nanos = index.to_numpy(dtype="datetime64[ns]").astype("int64")
|
| 39 |
+
deltas = np.diff(nanos) / 1e9 # seconds
|
| 40 |
+
deltas = deltas[deltas > 0]
|
| 41 |
+
if deltas.size == 0:
|
| 42 |
+
return _TRADING_DAYS
|
| 43 |
+
step = float(np.median(deltas))
|
| 44 |
+
if step >= 20 * 3600: # daily or slower -> use trading-day convention
|
| 45 |
+
days = step / 86400.0
|
| 46 |
+
return max(1, int(round(_TRADING_DAYS / max(days / 1.4, 1.0))))
|
| 47 |
+
# Intraday: assume a 6.5h session, 252 days a year.
|
| 48 |
+
bars_per_session = (6.5 * 3600) / step
|
| 49 |
+
return max(1, int(round(bars_per_session * _TRADING_DAYS)))
|
| 50 |
+
|
| 51 |
+
|
| 52 |
+
def _clean(returns: pd.Series) -> np.ndarray:
|
| 53 |
+
arr = np.asarray(returns, dtype=float)
|
| 54 |
+
return arr[np.isfinite(arr)]
|
| 55 |
+
|
| 56 |
+
|
| 57 |
+
def sharpe_ratio(returns: pd.Series, periods_per_year: int, rf: float = 0.0) -> float:
|
| 58 |
+
"""Annualised Sharpe. ``rf`` is an annual risk-free rate."""
|
| 59 |
+
arr = _clean(returns)
|
| 60 |
+
if arr.size < 2:
|
| 61 |
+
return 0.0
|
| 62 |
+
excess = arr - rf / periods_per_year
|
| 63 |
+
sd = excess.std(ddof=1)
|
| 64 |
+
if not np.isfinite(sd) or sd < _DEGENERATE_SD:
|
| 65 |
+
return 0.0
|
| 66 |
+
return float(excess.mean() / sd * np.sqrt(periods_per_year))
|
| 67 |
+
|
| 68 |
+
|
| 69 |
+
def sortino_ratio(returns: pd.Series, periods_per_year: int, rf: float = 0.0) -> float:
|
| 70 |
+
arr = _clean(returns)
|
| 71 |
+
if arr.size < 2:
|
| 72 |
+
return 0.0
|
| 73 |
+
excess = arr - rf / periods_per_year
|
| 74 |
+
downside = excess[excess < 0]
|
| 75 |
+
if downside.size == 0:
|
| 76 |
+
return float("inf") if excess.mean() > 0 else 0.0
|
| 77 |
+
dd = np.sqrt(np.mean(downside**2))
|
| 78 |
+
if not np.isfinite(dd) or dd < _DEGENERATE_SD:
|
| 79 |
+
return 0.0
|
| 80 |
+
return float(excess.mean() / dd * np.sqrt(periods_per_year))
|
| 81 |
+
|
| 82 |
+
|
| 83 |
+
def drawdown_series(equity: pd.Series) -> pd.Series:
|
| 84 |
+
peak = equity.cummax()
|
| 85 |
+
return equity / peak - 1.0
|
| 86 |
+
|
| 87 |
+
|
| 88 |
+
def max_drawdown(equity: pd.Series) -> float:
|
| 89 |
+
if equity.empty:
|
| 90 |
+
return 0.0
|
| 91 |
+
return float(drawdown_series(equity).min())
|
| 92 |
+
|
| 93 |
+
|
| 94 |
+
def _time_under_water(equity: pd.Series, periods_per_year: int) -> float:
|
| 95 |
+
"""Longest stretch below a prior peak, in years."""
|
| 96 |
+
if equity.empty:
|
| 97 |
+
return 0.0
|
| 98 |
+
dd = drawdown_series(equity).to_numpy()
|
| 99 |
+
longest = current = 0
|
| 100 |
+
for value in dd:
|
| 101 |
+
current = current + 1 if value < 0 else 0
|
| 102 |
+
longest = max(longest, current)
|
| 103 |
+
return longest / periods_per_year
|
| 104 |
+
|
| 105 |
+
|
| 106 |
+
def compute_metrics(
|
| 107 |
+
returns: pd.Series,
|
| 108 |
+
equity: pd.Series,
|
| 109 |
+
position: pd.Series | None = None,
|
| 110 |
+
periods_per_year: int | None = None,
|
| 111 |
+
rf: float = 0.0,
|
| 112 |
+
) -> Dict[str, float]:
|
| 113 |
+
"""Full metric bundle for one equity curve."""
|
| 114 |
+
ppy = periods_per_year or infer_periods_per_year(returns.index)
|
| 115 |
+
arr = _clean(returns)
|
| 116 |
+
n = arr.size
|
| 117 |
+
if n == 0 or equity.empty:
|
| 118 |
+
return {"periods_per_year": float(ppy)}
|
| 119 |
+
|
| 120 |
+
years = n / ppy
|
| 121 |
+
total_return = float(equity.iloc[-1] / equity.iloc[0] - 1.0)
|
| 122 |
+
cagr = float((equity.iloc[-1] / equity.iloc[0]) ** (1.0 / years) - 1.0) if years > 0 else 0.0
|
| 123 |
+
vol = float(arr.std(ddof=1) * np.sqrt(ppy))
|
| 124 |
+
mdd = max_drawdown(equity)
|
| 125 |
+
sr = sharpe_ratio(returns, ppy, rf)
|
| 126 |
+
|
| 127 |
+
out: Dict[str, float] = {
|
| 128 |
+
"total_return": total_return,
|
| 129 |
+
"cagr": cagr,
|
| 130 |
+
"ann_vol": vol,
|
| 131 |
+
"sharpe": sr,
|
| 132 |
+
"sortino": sortino_ratio(returns, ppy, rf),
|
| 133 |
+
"calmar": float(cagr / abs(mdd)) if mdd < 0 else 0.0,
|
| 134 |
+
"max_drawdown": mdd,
|
| 135 |
+
"time_under_water_yrs": _time_under_water(equity, ppy),
|
| 136 |
+
"hit_rate": float((arr > 0).mean()),
|
| 137 |
+
"skew": float(pd.Series(arr).skew()) if n > 2 else 0.0,
|
| 138 |
+
"kurtosis": float(pd.Series(arr).kurtosis()) if n > 3 else 0.0,
|
| 139 |
+
"var_95": float(np.percentile(arr, 5)),
|
| 140 |
+
"cvar_95": float(arr[arr <= np.percentile(arr, 5)].mean()) if n > 20 else 0.0,
|
| 141 |
+
"best_bar": float(arr.max()),
|
| 142 |
+
"worst_bar": float(arr.min()),
|
| 143 |
+
"n_bars": float(n),
|
| 144 |
+
"years": float(years),
|
| 145 |
+
"periods_per_year": float(ppy),
|
| 146 |
+
}
|
| 147 |
+
|
| 148 |
+
if position is not None and not position.empty:
|
| 149 |
+
pos = position.fillna(0.0)
|
| 150 |
+
turnover = pos.diff().abs().fillna(pos.abs().iloc[0] if len(pos) else 0.0)
|
| 151 |
+
out["exposure"] = float(pos.abs().mean())
|
| 152 |
+
out["long_share"] = float((pos > 0).mean())
|
| 153 |
+
out["short_share"] = float((pos < 0).mean())
|
| 154 |
+
out["turnover_ann"] = float(turnover.sum() / years) if years > 0 else 0.0
|
| 155 |
+
# A "trade" is any change in sign or size of exposure.
|
| 156 |
+
out["n_trades"] = float((turnover > 1e-9).sum())
|
| 157 |
+
return out
|
algotrader/panel.py
ADDED
|
@@ -0,0 +1,283 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Multi-asset price panels.
|
| 2 |
+
|
| 3 |
+
A :class:`Panel` is a set of aligned ``T x N`` frames -- one per OHLCV field,
|
| 4 |
+
one column per symbol. That is the shape cross-sectional work actually needs,
|
| 5 |
+
and it is what the portfolio engine consumes.
|
| 6 |
+
|
| 7 |
+
The important design choice here is that **missing data stays missing**. It is
|
| 8 |
+
tempting to forward-fill a symbol through the days it did not trade, but that
|
| 9 |
+
invents liquidity that never existed and quietly lets a strategy hold a
|
| 10 |
+
delisted stock forever. Instead the panel tracks exactly when each symbol was
|
| 11 |
+
tradable, which is also what makes survivorship measurable rather than assumed.
|
| 12 |
+
"""
|
| 13 |
+
|
| 14 |
+
from __future__ import annotations
|
| 15 |
+
|
| 16 |
+
from dataclasses import dataclass, field
|
| 17 |
+
from typing import Dict, Iterable, List, Mapping, Optional, Sequence
|
| 18 |
+
|
| 19 |
+
import numpy as np
|
| 20 |
+
import pandas as pd
|
| 21 |
+
|
| 22 |
+
from .data import load_ohlcv
|
| 23 |
+
from .types import OHLCV_COLUMNS
|
| 24 |
+
|
| 25 |
+
__all__ = ["Panel", "load_panel", "SurvivorshipReport"]
|
| 26 |
+
|
| 27 |
+
|
| 28 |
+
@dataclass(frozen=True)
|
| 29 |
+
class SurvivorshipReport:
|
| 30 |
+
"""How much of this universe is made of winners we already know survived."""
|
| 31 |
+
|
| 32 |
+
n_symbols: int
|
| 33 |
+
n_alive_at_end: int
|
| 34 |
+
n_delisted: int
|
| 35 |
+
delisted_symbols: List[str]
|
| 36 |
+
late_starters: List[str]
|
| 37 |
+
survival_rate: float
|
| 38 |
+
biased: bool
|
| 39 |
+
note: str
|
| 40 |
+
|
| 41 |
+
def as_flag(self) -> Optional[str]:
|
| 42 |
+
return self.note if self.biased else None
|
| 43 |
+
|
| 44 |
+
|
| 45 |
+
@dataclass(frozen=True)
|
| 46 |
+
class Panel:
|
| 47 |
+
"""Aligned multi-asset OHLCV."""
|
| 48 |
+
|
| 49 |
+
fields: Mapping[str, pd.DataFrame]
|
| 50 |
+
sources: Mapping[str, str] = field(default_factory=dict)
|
| 51 |
+
interval: str = "1d"
|
| 52 |
+
note: str = ""
|
| 53 |
+
|
| 54 |
+
def __post_init__(self) -> None:
|
| 55 |
+
missing = [c for c in OHLCV_COLUMNS if c not in self.fields]
|
| 56 |
+
if missing:
|
| 57 |
+
raise ValueError(f"Panel is missing field(s): {', '.join(missing)}")
|
| 58 |
+
reference = self.fields["close"]
|
| 59 |
+
for name, frame in self.fields.items():
|
| 60 |
+
if not frame.index.equals(reference.index) or list(frame.columns) != list(reference.columns):
|
| 61 |
+
raise ValueError(f"Panel field '{name}' is not aligned with 'close'")
|
| 62 |
+
|
| 63 |
+
# -- accessors ---------------------------------------------------------
|
| 64 |
+
@property
|
| 65 |
+
def close(self) -> pd.DataFrame:
|
| 66 |
+
return self.fields["close"]
|
| 67 |
+
|
| 68 |
+
@property
|
| 69 |
+
def open(self) -> pd.DataFrame:
|
| 70 |
+
return self.fields["open"]
|
| 71 |
+
|
| 72 |
+
@property
|
| 73 |
+
def high(self) -> pd.DataFrame:
|
| 74 |
+
return self.fields["high"]
|
| 75 |
+
|
| 76 |
+
@property
|
| 77 |
+
def low(self) -> pd.DataFrame:
|
| 78 |
+
return self.fields["low"]
|
| 79 |
+
|
| 80 |
+
@property
|
| 81 |
+
def volume(self) -> pd.DataFrame:
|
| 82 |
+
return self.fields["volume"]
|
| 83 |
+
|
| 84 |
+
@property
|
| 85 |
+
def symbols(self) -> List[str]:
|
| 86 |
+
return list(self.close.columns)
|
| 87 |
+
|
| 88 |
+
@property
|
| 89 |
+
def index(self) -> pd.DatetimeIndex:
|
| 90 |
+
return self.close.index
|
| 91 |
+
|
| 92 |
+
@property
|
| 93 |
+
def is_real(self) -> bool:
|
| 94 |
+
return all(s in ("yfinance", "bundled") for s in self.sources.values())
|
| 95 |
+
|
| 96 |
+
def __len__(self) -> int:
|
| 97 |
+
return len(self.close)
|
| 98 |
+
|
| 99 |
+
@property
|
| 100 |
+
def shape(self) -> tuple:
|
| 101 |
+
return self.close.shape
|
| 102 |
+
|
| 103 |
+
# -- derived -----------------------------------------------------------
|
| 104 |
+
def returns(self) -> pd.DataFrame:
|
| 105 |
+
"""Per-asset close-to-close returns, NaN where the asset was untradable."""
|
| 106 |
+
rets = self.close.pct_change()
|
| 107 |
+
return rets.where(self.tradable())
|
| 108 |
+
|
| 109 |
+
def tradable(self) -> pd.DataFrame:
|
| 110 |
+
"""True where the asset had a price on this bar *and* the one before.
|
| 111 |
+
|
| 112 |
+
A position can only be held over a bar whose return is defined, so this
|
| 113 |
+
is the mask the engine uses to zero out impossible weights.
|
| 114 |
+
"""
|
| 115 |
+
listed = self.close.notna()
|
| 116 |
+
return listed & listed.shift(1, fill_value=False)
|
| 117 |
+
|
| 118 |
+
def dollar_volume(self) -> pd.DataFrame:
|
| 119 |
+
return (self.close * self.volume).where(self.close.notna())
|
| 120 |
+
|
| 121 |
+
def first_valid(self) -> pd.Series:
|
| 122 |
+
return self.close.apply(lambda col: col.first_valid_index())
|
| 123 |
+
|
| 124 |
+
def last_valid(self) -> pd.Series:
|
| 125 |
+
return self.close.apply(lambda col: col.last_valid_index())
|
| 126 |
+
|
| 127 |
+
# -- survivorship ------------------------------------------------------
|
| 128 |
+
def survivorship(self, tolerance_bars: int = 5) -> SurvivorshipReport:
|
| 129 |
+
"""Measure how many names survived to the end of the sample.
|
| 130 |
+
|
| 131 |
+
A universe picked today and backfilled contains only survivors, and
|
| 132 |
+
every backtest run on it is flattered by the companies that failed and
|
| 133 |
+
were quietly excluded. We cannot fix that here, but we can refuse to
|
| 134 |
+
hide it: if every single name is still trading at the end of a long
|
| 135 |
+
sample, that is itself the evidence.
|
| 136 |
+
"""
|
| 137 |
+
if not len(self):
|
| 138 |
+
return SurvivorshipReport(0, 0, 0, [], [], 1.0, False, "Empty panel.")
|
| 139 |
+
|
| 140 |
+
last = self.last_valid()
|
| 141 |
+
first = self.first_valid()
|
| 142 |
+
end = self.index[-1]
|
| 143 |
+
start = self.index[0]
|
| 144 |
+
cutoff = self.index[max(0, len(self) - 1 - tolerance_bars)]
|
| 145 |
+
entry_cutoff = self.index[min(len(self) - 1, tolerance_bars)]
|
| 146 |
+
|
| 147 |
+
delisted = sorted(str(s) for s in last.index[last < cutoff])
|
| 148 |
+
late = sorted(str(s) for s in first.index[first > entry_cutoff])
|
| 149 |
+
n = len(self.symbols)
|
| 150 |
+
alive = n - len(delisted)
|
| 151 |
+
rate = alive / n if n else 1.0
|
| 152 |
+
|
| 153 |
+
years = len(self) / 252.0
|
| 154 |
+
biased = rate >= 1.0 and years >= 3 and n >= 5
|
| 155 |
+
if biased:
|
| 156 |
+
note = (
|
| 157 |
+
f"All {n} symbols were still trading at the end of a {years:.1f}-year sample. "
|
| 158 |
+
"A universe with no failures in it was almost certainly chosen after the fact, "
|
| 159 |
+
"which means these results exclude every name that went to zero. Treat the "
|
| 160 |
+
"returns below as an upper bound."
|
| 161 |
+
)
|
| 162 |
+
elif n == 0:
|
| 163 |
+
note = "Empty panel."
|
| 164 |
+
else:
|
| 165 |
+
note = (
|
| 166 |
+
f"{len(delisted)} of {n} symbols stopped trading before the end of the sample "
|
| 167 |
+
f"({rate:.0%} survived), so the universe is not made purely of winners."
|
| 168 |
+
)
|
| 169 |
+
|
| 170 |
+
return SurvivorshipReport(
|
| 171 |
+
n_symbols=n,
|
| 172 |
+
n_alive_at_end=alive,
|
| 173 |
+
n_delisted=len(delisted),
|
| 174 |
+
delisted_symbols=delisted[:25],
|
| 175 |
+
late_starters=late[:25],
|
| 176 |
+
survival_rate=float(rate),
|
| 177 |
+
biased=bool(biased),
|
| 178 |
+
note=note,
|
| 179 |
+
)
|
| 180 |
+
|
| 181 |
+
# -- construction ------------------------------------------------------
|
| 182 |
+
@classmethod
|
| 183 |
+
def from_frames(
|
| 184 |
+
cls,
|
| 185 |
+
frames: Mapping[str, pd.DataFrame],
|
| 186 |
+
sources: Optional[Mapping[str, str]] = None,
|
| 187 |
+
interval: str = "1d",
|
| 188 |
+
note: str = "",
|
| 189 |
+
min_bars: int = 2,
|
| 190 |
+
) -> "Panel":
|
| 191 |
+
"""Build a panel from ``{symbol: ohlcv_frame}``, aligning on the union index."""
|
| 192 |
+
usable = {
|
| 193 |
+
str(symbol): frame
|
| 194 |
+
for symbol, frame in frames.items()
|
| 195 |
+
if frame is not None and len(frame) >= min_bars
|
| 196 |
+
}
|
| 197 |
+
if not usable:
|
| 198 |
+
raise ValueError("No symbol had enough data to build a panel")
|
| 199 |
+
|
| 200 |
+
index = pd.DatetimeIndex([])
|
| 201 |
+
for frame in usable.values():
|
| 202 |
+
index = index.union(pd.DatetimeIndex(frame.index))
|
| 203 |
+
index = index.sort_values()
|
| 204 |
+
|
| 205 |
+
fields: Dict[str, pd.DataFrame] = {}
|
| 206 |
+
for column in OHLCV_COLUMNS:
|
| 207 |
+
fields[column] = pd.DataFrame(
|
| 208 |
+
{
|
| 209 |
+
symbol: pd.to_numeric(frame[column], errors="coerce").reindex(index)
|
| 210 |
+
for symbol, frame in usable.items()
|
| 211 |
+
},
|
| 212 |
+
index=index,
|
| 213 |
+
)
|
| 214 |
+
|
| 215 |
+
return cls(
|
| 216 |
+
fields=fields,
|
| 217 |
+
sources=dict(sources or {s: "unknown" for s in usable}),
|
| 218 |
+
interval=interval,
|
| 219 |
+
note=note,
|
| 220 |
+
)
|
| 221 |
+
|
| 222 |
+
def select(self, symbols: Sequence[str]) -> "Panel":
|
| 223 |
+
keep = [s for s in symbols if s in self.close.columns]
|
| 224 |
+
if not keep:
|
| 225 |
+
raise ValueError("None of the requested symbols are in this panel")
|
| 226 |
+
return Panel(
|
| 227 |
+
fields={name: frame.loc[:, keep] for name, frame in self.fields.items()},
|
| 228 |
+
sources={s: self.sources.get(s, "unknown") for s in keep},
|
| 229 |
+
interval=self.interval,
|
| 230 |
+
note=self.note,
|
| 231 |
+
)
|
| 232 |
+
|
| 233 |
+
def slice(self, start=None, end=None) -> "Panel":
|
| 234 |
+
return Panel(
|
| 235 |
+
fields={name: frame.loc[start:end] for name, frame in self.fields.items()},
|
| 236 |
+
sources=dict(self.sources),
|
| 237 |
+
interval=self.interval,
|
| 238 |
+
note=self.note,
|
| 239 |
+
)
|
| 240 |
+
|
| 241 |
+
|
| 242 |
+
def load_panel(
|
| 243 |
+
symbols: Iterable[str],
|
| 244 |
+
start: str = "2015-01-01",
|
| 245 |
+
end: Optional[str] = None,
|
| 246 |
+
interval: str = "1d",
|
| 247 |
+
source: str = "auto",
|
| 248 |
+
min_bars: int = 120,
|
| 249 |
+
) -> Panel:
|
| 250 |
+
"""Load a panel for ``symbols``, skipping any that cannot supply enough history."""
|
| 251 |
+
symbols = [str(s).strip().upper() for s in symbols if str(s).strip()]
|
| 252 |
+
if not symbols:
|
| 253 |
+
raise ValueError("No symbols requested")
|
| 254 |
+
|
| 255 |
+
frames: Dict[str, pd.DataFrame] = {}
|
| 256 |
+
sources: Dict[str, str] = {}
|
| 257 |
+
skipped: List[str] = []
|
| 258 |
+
|
| 259 |
+
for symbol in dict.fromkeys(symbols): # de-duplicate, keep order
|
| 260 |
+
market = load_ohlcv(symbol, start, end, interval, source)
|
| 261 |
+
if len(market.df) < min_bars:
|
| 262 |
+
skipped.append(symbol)
|
| 263 |
+
continue
|
| 264 |
+
frames[symbol] = market.df
|
| 265 |
+
sources[symbol] = market.source
|
| 266 |
+
|
| 267 |
+
if not frames:
|
| 268 |
+
raise ValueError(
|
| 269 |
+
f"None of {len(symbols)} symbols returned at least {min_bars} bars."
|
| 270 |
+
)
|
| 271 |
+
|
| 272 |
+
simulated = sorted(s for s, src in sources.items() if src == "synthetic")
|
| 273 |
+
note = ""
|
| 274 |
+
if simulated:
|
| 275 |
+
note = (
|
| 276 |
+
f"{len(simulated)} of {len(frames)} symbols fell back to the market simulator "
|
| 277 |
+
f"({', '.join(simulated[:6])}{'...' if len(simulated) > 6 else ''}). "
|
| 278 |
+
"The statistics are still valid; they are measured on a simulated market."
|
| 279 |
+
)
|
| 280 |
+
if skipped:
|
| 281 |
+
note = (note + " " if note else "") + f"Skipped for insufficient history: {', '.join(skipped[:6])}."
|
| 282 |
+
|
| 283 |
+
return Panel.from_frames(frames, sources, interval=interval, note=note.strip())
|
algotrader/portfolio.py
ADDED
|
@@ -0,0 +1,257 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Matrix portfolio engine.
|
| 2 |
+
|
| 3 |
+
The single-asset engine treats turnover as ``|target[t] - target[t-1]|``. That
|
| 4 |
+
is wrong the moment weights are fractional, because a position you did not
|
| 5 |
+
touch still *drifts*: hold 50% of your book in a name that doubles and you are
|
| 6 |
+
now at 67% without trading. Charging costs against the previous target instead
|
| 7 |
+
of the previous *actual* weight understates the cost of doing nothing and
|
| 8 |
+
overstates the cost of rebalancing.
|
| 9 |
+
|
| 10 |
+
This module models the drift explicitly and closes the gap:
|
| 11 |
+
|
| 12 |
+
w_start[t] = target[t - lag] what we want to hold
|
| 13 |
+
r_p[t] = sum(w_start[t] * R[t]) portfolio return that bar
|
| 14 |
+
w_end[t] = w_start[t] * (1 + R[t]) / (1 + r_p[t]) drifted by the bar
|
| 15 |
+
turnover[t] = sum |w_start[t] - w_end[t - 1]| what we actually traded
|
| 16 |
+
|
| 17 |
+
Every step is a function of the current bar and the one before, so the whole
|
| 18 |
+
thing stays vectorised -- no Python loop over time.
|
| 19 |
+
"""
|
| 20 |
+
|
| 21 |
+
from __future__ import annotations
|
| 22 |
+
|
| 23 |
+
from typing import Optional
|
| 24 |
+
|
| 25 |
+
import numpy as np
|
| 26 |
+
import pandas as pd
|
| 27 |
+
|
| 28 |
+
from .metrics import compute_metrics, infer_periods_per_year
|
| 29 |
+
from .panel import Panel
|
| 30 |
+
from .types import BacktestResult, CostModel
|
| 31 |
+
|
| 32 |
+
__all__ = ["run_portfolio_backtest", "PortfolioResult", "normalise_weights"]
|
| 33 |
+
|
| 34 |
+
|
| 35 |
+
class PortfolioResult(BacktestResult):
|
| 36 |
+
"""A :class:`BacktestResult` that also keeps the per-asset weight history."""
|
| 37 |
+
|
| 38 |
+
def __init__(self, *args, weights: pd.DataFrame, held: pd.DataFrame, panel: Panel, **kwargs):
|
| 39 |
+
super().__init__(*args, **kwargs)
|
| 40 |
+
self.weights = weights # requested, post-constraint
|
| 41 |
+
self.held = held # actually held during each bar
|
| 42 |
+
self.panel = panel
|
| 43 |
+
|
| 44 |
+
def attribution(self) -> pd.Series:
|
| 45 |
+
"""Total return contribution per symbol, largest first."""
|
| 46 |
+
contrib = (self.held * self.panel.returns().fillna(0.0)).sum(axis=0)
|
| 47 |
+
return contrib.sort_values(ascending=False)
|
| 48 |
+
|
| 49 |
+
|
| 50 |
+
def normalise_weights(
|
| 51 |
+
weights: pd.DataFrame,
|
| 52 |
+
listed: pd.DataFrame,
|
| 53 |
+
gross_leverage: float = 1.0,
|
| 54 |
+
max_weight: Optional[float] = None,
|
| 55 |
+
allow_short: bool = True,
|
| 56 |
+
) -> pd.DataFrame:
|
| 57 |
+
"""Apply the constraints a real book has, in the order a real book applies them.
|
| 58 |
+
|
| 59 |
+
``listed`` is "did this name have a price when the decision was made" --
|
| 60 |
+
deliberately not the stricter "is a return defined over this bar" mask. A
|
| 61 |
+
decision taken at Monday's close only needs Monday's price to exist; whether
|
| 62 |
+
the position can actually be carried is enforced after the lag shift.
|
| 63 |
+
"""
|
| 64 |
+
w = weights.reindex(index=listed.index, columns=listed.columns).astype(float).fillna(0.0)
|
| 65 |
+
|
| 66 |
+
# You cannot ask for exposure to something that is not listed yet.
|
| 67 |
+
w = w.where(listed, 0.0)
|
| 68 |
+
|
| 69 |
+
if not allow_short:
|
| 70 |
+
w = w.clip(lower=0.0)
|
| 71 |
+
if max_weight is not None:
|
| 72 |
+
w = w.clip(-abs(max_weight), abs(max_weight))
|
| 73 |
+
|
| 74 |
+
# Scale down (never up) so gross exposure respects the leverage cap.
|
| 75 |
+
gross = w.abs().sum(axis=1)
|
| 76 |
+
scale = np.minimum(1.0, gross_leverage / gross.replace(0.0, np.nan))
|
| 77 |
+
return w.mul(scale.fillna(1.0), axis=0)
|
| 78 |
+
|
| 79 |
+
|
| 80 |
+
def run_portfolio_backtest(
|
| 81 |
+
panel: Panel,
|
| 82 |
+
weights: pd.DataFrame,
|
| 83 |
+
costs: Optional[CostModel] = None,
|
| 84 |
+
lag: int = 1,
|
| 85 |
+
gross_leverage: float = 1.0,
|
| 86 |
+
max_weight: Optional[float] = None,
|
| 87 |
+
allow_short: bool = True,
|
| 88 |
+
initial_capital: float = 100_000.0,
|
| 89 |
+
periods_per_year: Optional[int] = None,
|
| 90 |
+
rf: float = 0.0,
|
| 91 |
+
benchmark: Optional[pd.Series] = None,
|
| 92 |
+
rebalance_on: Optional[pd.Series] = None,
|
| 93 |
+
meta: Optional[dict] = None,
|
| 94 |
+
) -> PortfolioResult:
|
| 95 |
+
"""Backtest a ``T x N`` weight matrix against a panel.
|
| 96 |
+
|
| 97 |
+
``weights[t]`` is the exposure decided using information up to the close of
|
| 98 |
+
bar ``t``; it is held from bar ``t + lag``. The default benchmark is the
|
| 99 |
+
equal-weight universe, which is a far more honest comparison for a
|
| 100 |
+
cross-sectional strategy than any single ticker.
|
| 101 |
+
"""
|
| 102 |
+
if len(panel) < 2:
|
| 103 |
+
raise ValueError("Cannot backtest a panel with fewer than two bars")
|
| 104 |
+
if lag < 1:
|
| 105 |
+
raise ValueError("lag must be >= 1; lag=0 would trade on unavailable information")
|
| 106 |
+
|
| 107 |
+
costs = costs or CostModel()
|
| 108 |
+
ppy = periods_per_year or infer_periods_per_year(panel.index)
|
| 109 |
+
|
| 110 |
+
asset_returns = panel.returns().fillna(0.0)
|
| 111 |
+
tradable = panel.tradable()
|
| 112 |
+
listed = panel.close.notna()
|
| 113 |
+
|
| 114 |
+
target = normalise_weights(weights, listed, gross_leverage, max_weight, allow_short)
|
| 115 |
+
held = target.shift(lag).fillna(0.0)
|
| 116 |
+
# Re-apply tradability after the shift: a name can delist between the
|
| 117 |
+
# decision and the fill, and we must not be holding it when it does.
|
| 118 |
+
held = held.where(tradable, 0.0)
|
| 119 |
+
|
| 120 |
+
if rebalance_on is not None:
|
| 121 |
+
held = _apply_rebalance_schedule(held, asset_returns, tradable, rebalance_on)
|
| 122 |
+
|
| 123 |
+
gross_return = (held * asset_returns).sum(axis=1)
|
| 124 |
+
|
| 125 |
+
# Weights after the bar's move, renormalised to the new portfolio value.
|
| 126 |
+
growth = (1.0 + gross_return).replace(0.0, np.nan)
|
| 127 |
+
drifted = (held * (1.0 + asset_returns)).div(growth, axis=0).fillna(0.0)
|
| 128 |
+
previous = drifted.shift(1).fillna(0.0)
|
| 129 |
+
|
| 130 |
+
traded = (held - previous).abs().sum(axis=1)
|
| 131 |
+
trade_cost = traded * (costs.one_way_bps / 1e4)
|
| 132 |
+
borrow_cost = held.clip(upper=0.0).abs().sum(axis=1) * (costs.short_borrow_bps / 1e4) / ppy
|
| 133 |
+
total_cost = trade_cost + borrow_cost
|
| 134 |
+
|
| 135 |
+
net = gross_return - total_cost
|
| 136 |
+
equity = initial_capital * (1.0 + net).cumprod()
|
| 137 |
+
|
| 138 |
+
if benchmark is None:
|
| 139 |
+
# Equal weight across whatever was tradable on each bar.
|
| 140 |
+
counts = tradable.sum(axis=1).replace(0, np.nan)
|
| 141 |
+
equal = tradable.astype(float).div(counts, axis=0).fillna(0.0)
|
| 142 |
+
benchmark = (equal.shift(lag).fillna(0.0) * asset_returns).sum(axis=1)
|
| 143 |
+
benchmark = benchmark.reindex(panel.index).fillna(0.0)
|
| 144 |
+
benchmark_equity = initial_capital * (1.0 + benchmark).cumprod()
|
| 145 |
+
|
| 146 |
+
exposure = held.abs().sum(axis=1)
|
| 147 |
+
metrics = compute_metrics(net, equity, exposure, ppy, rf)
|
| 148 |
+
metrics.update(_portfolio_metrics(held, traded, metrics.get("years", 1.0)))
|
| 149 |
+
|
| 150 |
+
survivorship = panel.survivorship()
|
| 151 |
+
|
| 152 |
+
result = PortfolioResult(
|
| 153 |
+
equity=equity,
|
| 154 |
+
returns=net,
|
| 155 |
+
gross_returns=gross_return,
|
| 156 |
+
position=exposure,
|
| 157 |
+
target=target.abs().sum(axis=1),
|
| 158 |
+
costs=total_cost,
|
| 159 |
+
benchmark_equity=benchmark_equity,
|
| 160 |
+
metrics=metrics,
|
| 161 |
+
benchmark_metrics=compute_metrics(benchmark, benchmark_equity, None, ppy, rf),
|
| 162 |
+
meta={
|
| 163 |
+
"lag": lag,
|
| 164 |
+
"gross_leverage": gross_leverage,
|
| 165 |
+
"max_weight": max_weight,
|
| 166 |
+
"allow_short": allow_short,
|
| 167 |
+
"initial_capital": initial_capital,
|
| 168 |
+
"periods_per_year": ppy,
|
| 169 |
+
"n_symbols": len(panel.symbols),
|
| 170 |
+
"survivorship": survivorship,
|
| 171 |
+
**(meta or {}),
|
| 172 |
+
},
|
| 173 |
+
weights=target,
|
| 174 |
+
held=held,
|
| 175 |
+
panel=panel,
|
| 176 |
+
)
|
| 177 |
+
return result
|
| 178 |
+
|
| 179 |
+
|
| 180 |
+
def _apply_rebalance_schedule(
|
| 181 |
+
held: pd.DataFrame,
|
| 182 |
+
asset_returns: pd.DataFrame,
|
| 183 |
+
tradable: pd.DataFrame,
|
| 184 |
+
rebalance_on: pd.Series,
|
| 185 |
+
) -> pd.DataFrame:
|
| 186 |
+
"""Trade only on rebalance bars; let the book drift in between.
|
| 187 |
+
|
| 188 |
+
Without this, a monthly strategy whose target is constant between
|
| 189 |
+
rebalances gets charged turnover every single bar for holding still --
|
| 190 |
+
the model would be paying to *prevent* drift that a real book simply lets
|
| 191 |
+
happen. This is the one genuinely recursive step in the engine: today's
|
| 192 |
+
holding depends on yesterday's drifted holding.
|
| 193 |
+
"""
|
| 194 |
+
schedule = rebalance_on.reindex(held.index).fillna(False).to_numpy(dtype=bool)
|
| 195 |
+
target = held.to_numpy(dtype=float)
|
| 196 |
+
returns = asset_returns.to_numpy(dtype=float)
|
| 197 |
+
can_hold = tradable.to_numpy(dtype=bool)
|
| 198 |
+
|
| 199 |
+
n_bars, n_assets = target.shape
|
| 200 |
+
out = np.zeros((n_bars, n_assets), dtype=float)
|
| 201 |
+
carried = np.zeros(n_assets, dtype=float)
|
| 202 |
+
|
| 203 |
+
for t in range(n_bars):
|
| 204 |
+
current = target[t] if schedule[t] else carried
|
| 205 |
+
current = np.where(can_hold[t], current, 0.0)
|
| 206 |
+
out[t] = current
|
| 207 |
+
# Drift into the next bar, renormalised to the new portfolio value.
|
| 208 |
+
portfolio_return = float(current @ returns[t])
|
| 209 |
+
growth = 1.0 + portfolio_return
|
| 210 |
+
carried = current * (1.0 + returns[t]) / growth if abs(growth) > 1e-12 else current
|
| 211 |
+
|
| 212 |
+
return pd.DataFrame(out, index=held.index, columns=held.columns)
|
| 213 |
+
|
| 214 |
+
|
| 215 |
+
def rebalance_schedule(index: pd.DatetimeIndex, frequency: str = "M") -> pd.Series:
|
| 216 |
+
"""Boolean per-bar mask marking rebalance dates.
|
| 217 |
+
|
| 218 |
+
``frequency`` is ``D`` (every bar), ``W``, ``M``, ``Q``, or an integer
|
| 219 |
+
number of bars as a string.
|
| 220 |
+
"""
|
| 221 |
+
frequency = str(frequency).upper().strip()
|
| 222 |
+
if frequency in ("D", "1", "B", ""):
|
| 223 |
+
return pd.Series(True, index=index)
|
| 224 |
+
if frequency.isdigit():
|
| 225 |
+
step = max(1, int(frequency))
|
| 226 |
+
mask = np.zeros(len(index), dtype=bool)
|
| 227 |
+
mask[::step] = True
|
| 228 |
+
return pd.Series(mask, index=index)
|
| 229 |
+
|
| 230 |
+
periods = {"W": index.to_period("W"), "M": index.to_period("M"), "Q": index.to_period("Q")}
|
| 231 |
+
if frequency not in periods:
|
| 232 |
+
raise ValueError(f"Unknown rebalance frequency '{frequency}'")
|
| 233 |
+
period = periods[frequency]
|
| 234 |
+
# First bar of each period -- known at the time, unlike the last bar.
|
| 235 |
+
return pd.Series(period != pd.Series(period, index=index).shift(1).to_numpy(), index=index)
|
| 236 |
+
|
| 237 |
+
|
| 238 |
+
def _portfolio_metrics(held: pd.DataFrame, traded: pd.Series, years: float) -> dict:
|
| 239 |
+
"""Book-level statistics a portfolio manager will look for first."""
|
| 240 |
+
absolute = held.abs()
|
| 241 |
+
gross = absolute.sum(axis=1)
|
| 242 |
+
active = (absolute > 1e-9).sum(axis=1)
|
| 243 |
+
|
| 244 |
+
# Herfindahl on the gross book: 1.0 is everything in one name, 1/n is even.
|
| 245 |
+
shares = absolute.div(gross.replace(0.0, np.nan), axis=0)
|
| 246 |
+
hhi = (shares**2).sum(axis=1)
|
| 247 |
+
|
| 248 |
+
return {
|
| 249 |
+
"gross_exposure": float(gross.mean()),
|
| 250 |
+
"net_exposure": float(held.sum(axis=1).mean()),
|
| 251 |
+
"max_gross_exposure": float(gross.max()),
|
| 252 |
+
"avg_positions": float(active.mean()),
|
| 253 |
+
"max_positions": float(active.max()),
|
| 254 |
+
"concentration_hhi": float(hhi.mean(skipna=True)) if hhi.notna().any() else float("nan"),
|
| 255 |
+
"turnover_ann": float(traded.sum() / years) if years > 0 else 0.0,
|
| 256 |
+
"n_trades": float((traded > 1e-9).sum()),
|
| 257 |
+
}
|
algotrader/portfolio_lab.py
ADDED
|
@@ -0,0 +1,319 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""The Portfolio Lab: the honesty pipeline for cross-sectional strategies.
|
| 2 |
+
|
| 3 |
+
Same idea as :mod:`algotrader.lab`, but the questions change when you move from
|
| 4 |
+
one asset to many. A timing rule has to prove the market had structure. A book
|
| 5 |
+
that ranks names has to prove three harder things:
|
| 6 |
+
|
| 7 |
+
1. it picked the right names (cross-sectional permutation);
|
| 8 |
+
2. what it picked is not just a style you could buy in an ETF (attribution);
|
| 9 |
+
3. the universe it picked from contains the losers as well as the winners
|
| 10 |
+
(survivorship).
|
| 11 |
+
|
| 12 |
+
All three are wired into the Reality Score alongside the usual selection-bias
|
| 13 |
+
and walk-forward machinery.
|
| 14 |
+
"""
|
| 15 |
+
|
| 16 |
+
from __future__ import annotations
|
| 17 |
+
|
| 18 |
+
import logging
|
| 19 |
+
from dataclasses import dataclass, field
|
| 20 |
+
from typing import Callable, Dict, List, Optional, Sequence
|
| 21 |
+
|
| 22 |
+
import numpy as np
|
| 23 |
+
import pandas as pd
|
| 24 |
+
|
| 25 |
+
from .attribution import build_style_factors, factor_attribution
|
| 26 |
+
from .cross_sectional import CrossSectionalStrategy, get_xs_strategy, list_xs_strategies
|
| 27 |
+
from .metrics import infer_periods_per_year
|
| 28 |
+
from .panel import Panel, load_panel
|
| 29 |
+
from .portfolio import PortfolioResult, rebalance_schedule, run_portfolio_backtest
|
| 30 |
+
from .types import CostModel
|
| 31 |
+
from .validation.cross_permutation import CrossPermutationResult, cross_sectional_permutation_test
|
| 32 |
+
from .validation.deflated_sharpe import deflated_sharpe_ratio
|
| 33 |
+
from .validation.pbo import probability_of_backtest_overfitting
|
| 34 |
+
from .validation.walkforward import walk_forward_panel
|
| 35 |
+
from .verdict import reality_score
|
| 36 |
+
|
| 37 |
+
logger = logging.getLogger(__name__)
|
| 38 |
+
|
| 39 |
+
__all__ = ["PortfolioLabConfig", "PortfolioLabReport", "run_portfolio_lab", "run_portfolio_arena"]
|
| 40 |
+
|
| 41 |
+
ProgressFn = Optional[Callable[[float, str], None]]
|
| 42 |
+
|
| 43 |
+
DEFAULT_UNIVERSE = [
|
| 44 |
+
"SPY", "QQQ", "AAPL", "MSFT", "NVDA", "AMZN",
|
| 45 |
+
"META", "TSLA", "GOOGL", "GLD", "TLT", "BTC-USD",
|
| 46 |
+
]
|
| 47 |
+
|
| 48 |
+
|
| 49 |
+
@dataclass
|
| 50 |
+
class PortfolioLabConfig:
|
| 51 |
+
symbols: Sequence[str] = tuple(DEFAULT_UNIVERSE)
|
| 52 |
+
start: str = "2015-01-01"
|
| 53 |
+
end: Optional[str] = None
|
| 54 |
+
interval: str = "1d"
|
| 55 |
+
source: str = "yahoo"
|
| 56 |
+
|
| 57 |
+
strategy: str = "xs_momentum"
|
| 58 |
+
params: Dict[str, float] = field(default_factory=dict)
|
| 59 |
+
|
| 60 |
+
commission_bps: float = 1.0
|
| 61 |
+
slippage_bps: float = 2.0
|
| 62 |
+
short_borrow_bps: float = 50.0
|
| 63 |
+
lag: int = 1
|
| 64 |
+
gross_leverage: float = 1.0
|
| 65 |
+
max_weight: Optional[float] = 0.25
|
| 66 |
+
allow_short: bool = True
|
| 67 |
+
rebalance: str = "M"
|
| 68 |
+
capital: float = 1_000_000.0
|
| 69 |
+
|
| 70 |
+
n_permutations: int = 150
|
| 71 |
+
wf_folds: int = 4
|
| 72 |
+
pbo_splits: int = 8
|
| 73 |
+
grid_limit: int = 24
|
| 74 |
+
seed: int = 0
|
| 75 |
+
|
| 76 |
+
def costs(self, multiplier: float = 1.0) -> CostModel:
|
| 77 |
+
return CostModel(
|
| 78 |
+
commission_bps=self.commission_bps * multiplier,
|
| 79 |
+
slippage_bps=self.slippage_bps * multiplier,
|
| 80 |
+
short_borrow_bps=self.short_borrow_bps * multiplier,
|
| 81 |
+
)
|
| 82 |
+
|
| 83 |
+
|
| 84 |
+
@dataclass
|
| 85 |
+
class PortfolioLabReport:
|
| 86 |
+
config: PortfolioLabConfig
|
| 87 |
+
panel: Panel
|
| 88 |
+
strategy: CrossSectionalStrategy
|
| 89 |
+
params: Dict[str, float]
|
| 90 |
+
backtest: PortfolioResult
|
| 91 |
+
permutation: Optional[CrossPermutationResult] = None
|
| 92 |
+
dsr: Dict[str, float] = field(default_factory=dict)
|
| 93 |
+
pbo: Dict[str, object] = field(default_factory=dict)
|
| 94 |
+
walkforward: Dict[str, object] = field(default_factory=dict)
|
| 95 |
+
attribution: Dict[str, object] = field(default_factory=dict)
|
| 96 |
+
trials: Dict[str, object] = field(default_factory=dict)
|
| 97 |
+
verdict: Dict[str, object] = field(default_factory=dict)
|
| 98 |
+
cost_stress: Dict[str, float] = field(default_factory=dict)
|
| 99 |
+
|
| 100 |
+
@property
|
| 101 |
+
def survivorship(self):
|
| 102 |
+
return self.panel.survivorship()
|
| 103 |
+
|
| 104 |
+
|
| 105 |
+
def _trial_matrix(
|
| 106 |
+
panel: Panel,
|
| 107 |
+
strategy: CrossSectionalStrategy,
|
| 108 |
+
cfg: PortfolioLabConfig,
|
| 109 |
+
schedule: pd.Series,
|
| 110 |
+
progress: ProgressFn = None,
|
| 111 |
+
) -> tuple[np.ndarray, List[float], List[str]]:
|
| 112 |
+
"""Backtest every parameter variant, for the Deflated Sharpe and PBO inputs."""
|
| 113 |
+
grid = strategy.grid(limit=cfg.grid_limit)
|
| 114 |
+
costs = cfg.costs()
|
| 115 |
+
columns, sharpes, labels = [], [], []
|
| 116 |
+
|
| 117 |
+
for i, params in enumerate(grid):
|
| 118 |
+
weights = strategy.generate(panel, params)
|
| 119 |
+
result = run_portfolio_backtest(
|
| 120 |
+
panel, weights, costs=costs, lag=cfg.lag,
|
| 121 |
+
gross_leverage=cfg.gross_leverage, max_weight=cfg.max_weight,
|
| 122 |
+
allow_short=cfg.allow_short, rebalance_on=schedule,
|
| 123 |
+
)
|
| 124 |
+
columns.append(result.returns.to_numpy(dtype=float))
|
| 125 |
+
sharpes.append(result.sharpe)
|
| 126 |
+
labels.append(", ".join(f"{k}={v}" for k, v in params.items()) or "default")
|
| 127 |
+
if progress is not None and i % 3 == 0:
|
| 128 |
+
progress((i + 1) / max(len(grid), 1), f"Variant {i + 1}/{len(grid)}")
|
| 129 |
+
|
| 130 |
+
matrix = np.column_stack(columns) if columns else np.zeros((len(panel), 0))
|
| 131 |
+
return matrix, sharpes, labels
|
| 132 |
+
|
| 133 |
+
|
| 134 |
+
def run_portfolio_lab(cfg: PortfolioLabConfig, progress: ProgressFn = None) -> PortfolioLabReport:
|
| 135 |
+
"""Run the full cross-sectional honesty pipeline."""
|
| 136 |
+
|
| 137 |
+
def step(fraction: float, message: str) -> None:
|
| 138 |
+
if progress is not None:
|
| 139 |
+
progress(min(max(fraction, 0.0), 1.0), message)
|
| 140 |
+
|
| 141 |
+
step(0.02, f"Loading {len(cfg.symbols)} symbols")
|
| 142 |
+
panel = load_panel(cfg.symbols, cfg.start, cfg.end, cfg.interval, cfg.source)
|
| 143 |
+
if len(panel) < 250:
|
| 144 |
+
raise ValueError(
|
| 145 |
+
f"Only {len(panel)} bars available. Widen the date range — a cross-sectional "
|
| 146 |
+
"book cannot be validated on less than a year of data."
|
| 147 |
+
)
|
| 148 |
+
|
| 149 |
+
strategy = get_xs_strategy(cfg.strategy)
|
| 150 |
+
params = strategy.clean(cfg.params)
|
| 151 |
+
ppy = infer_periods_per_year(panel.index)
|
| 152 |
+
schedule = rebalance_schedule(panel.index, cfg.rebalance)
|
| 153 |
+
|
| 154 |
+
step(0.10, "Running the backtest")
|
| 155 |
+
weights = strategy.generate(panel, params)
|
| 156 |
+
backtest = run_portfolio_backtest(
|
| 157 |
+
panel, weights, costs=cfg.costs(), lag=cfg.lag,
|
| 158 |
+
gross_leverage=cfg.gross_leverage, max_weight=cfg.max_weight,
|
| 159 |
+
allow_short=cfg.allow_short, initial_capital=cfg.capital,
|
| 160 |
+
periods_per_year=ppy, rebalance_on=schedule,
|
| 161 |
+
meta={"strategy": strategy.key, "params": params, "rebalance": cfg.rebalance},
|
| 162 |
+
)
|
| 163 |
+
|
| 164 |
+
step(0.16, "Stress-testing costs")
|
| 165 |
+
stressed = run_portfolio_backtest(
|
| 166 |
+
panel, weights, costs=cfg.costs(3.0), lag=cfg.lag,
|
| 167 |
+
gross_leverage=cfg.gross_leverage, max_weight=cfg.max_weight,
|
| 168 |
+
allow_short=cfg.allow_short, periods_per_year=ppy, rebalance_on=schedule,
|
| 169 |
+
)
|
| 170 |
+
base_sharpe = backtest.sharpe
|
| 171 |
+
cost_stress = {
|
| 172 |
+
"sharpe_1x": base_sharpe,
|
| 173 |
+
"sharpe_3x": stressed.sharpe,
|
| 174 |
+
"ratio": float(stressed.sharpe / base_sharpe) if base_sharpe > 1e-9 else 0.0,
|
| 175 |
+
"return_3x": float(stressed.metrics.get("total_return", 0.0)),
|
| 176 |
+
}
|
| 177 |
+
|
| 178 |
+
step(0.22, "Backtesting every parameter variant")
|
| 179 |
+
matrix, trial_sharpes, labels = _trial_matrix(
|
| 180 |
+
panel, strategy, cfg, schedule, lambda f, m: step(0.22 + 0.14 * f, m)
|
| 181 |
+
)
|
| 182 |
+
n_trials = max(len(trial_sharpes), 1)
|
| 183 |
+
|
| 184 |
+
step(0.38, "Deflating the Sharpe ratio for selection bias")
|
| 185 |
+
dsr = deflated_sharpe_ratio(
|
| 186 |
+
backtest.returns.to_numpy(dtype=float),
|
| 187 |
+
sharpe_annual=base_sharpe,
|
| 188 |
+
periods_per_year=ppy,
|
| 189 |
+
n_trials=n_trials,
|
| 190 |
+
trial_sharpes=trial_sharpes if n_trials > 1 else None,
|
| 191 |
+
)
|
| 192 |
+
|
| 193 |
+
step(0.42, "Measuring backtest overfitting")
|
| 194 |
+
pbo = probability_of_backtest_overfitting(matrix, n_splits=cfg.pbo_splits, labels=labels)
|
| 195 |
+
|
| 196 |
+
step(0.46, "Shuffling names within each date")
|
| 197 |
+
permutation = None
|
| 198 |
+
if cfg.n_permutations > 0:
|
| 199 |
+
permutation = cross_sectional_permutation_test(
|
| 200 |
+
panel, weights, n_permutations=cfg.n_permutations, lag=cfg.lag,
|
| 201 |
+
gross_leverage=cfg.gross_leverage, max_weight=cfg.max_weight,
|
| 202 |
+
allow_short=cfg.allow_short, rebalance_on=schedule, seed=cfg.seed,
|
| 203 |
+
progress=lambda f, m: step(0.46 + 0.30 * f, m),
|
| 204 |
+
)
|
| 205 |
+
|
| 206 |
+
step(0.78, "Attributing returns to style factors")
|
| 207 |
+
try:
|
| 208 |
+
factors = build_style_factors(panel)
|
| 209 |
+
attribution = factor_attribution(backtest.returns, factors, ppy)
|
| 210 |
+
except Exception as exc: # noqa: BLE001 - attribution must never sink a run
|
| 211 |
+
logger.warning("Attribution failed: %s", exc)
|
| 212 |
+
attribution = {"available": False, "note": f"Attribution unavailable: {exc}"}
|
| 213 |
+
|
| 214 |
+
step(0.86, "Walking the strategy forward")
|
| 215 |
+
wf = walk_forward_panel(
|
| 216 |
+
panel, strategy, n_folds=cfg.wf_folds, costs=cfg.costs(), lag=cfg.lag,
|
| 217 |
+
gross_leverage=cfg.gross_leverage, allow_short=cfg.allow_short,
|
| 218 |
+
rebalance=cfg.rebalance, grid_limit=min(cfg.grid_limit, 12),
|
| 219 |
+
progress=lambda f, m: step(0.86 + 0.10 * f, m),
|
| 220 |
+
)
|
| 221 |
+
|
| 222 |
+
step(0.98, "Grading")
|
| 223 |
+
survivorship = panel.survivorship()
|
| 224 |
+
verdict = reality_score(
|
| 225 |
+
metrics=backtest.metrics,
|
| 226 |
+
benchmark_metrics=backtest.benchmark_metrics,
|
| 227 |
+
p_value=permutation.p_value if permutation else None,
|
| 228 |
+
dsr=dsr.get("dsr"),
|
| 229 |
+
pbo=pbo.get("pbo"),
|
| 230 |
+
wf_efficiency=wf.get("efficiency"),
|
| 231 |
+
wf_win_rate=wf.get("oos_win_rate"),
|
| 232 |
+
cost_stress_ratio=cost_stress["ratio"],
|
| 233 |
+
attribution=attribution,
|
| 234 |
+
survivorship=survivorship,
|
| 235 |
+
benchmark_name="The equal-weight universe",
|
| 236 |
+
permutation_label="books with the same shape but randomly chosen names",
|
| 237 |
+
)
|
| 238 |
+
|
| 239 |
+
step(1.0, "Done")
|
| 240 |
+
return PortfolioLabReport(
|
| 241 |
+
config=cfg,
|
| 242 |
+
panel=panel,
|
| 243 |
+
strategy=strategy,
|
| 244 |
+
params=params,
|
| 245 |
+
backtest=backtest,
|
| 246 |
+
permutation=permutation,
|
| 247 |
+
dsr=dsr,
|
| 248 |
+
pbo=pbo,
|
| 249 |
+
walkforward=wf,
|
| 250 |
+
attribution=attribution,
|
| 251 |
+
trials={"n": n_trials, "sharpes": trial_sharpes, "labels": labels},
|
| 252 |
+
verdict=verdict,
|
| 253 |
+
cost_stress=cost_stress,
|
| 254 |
+
)
|
| 255 |
+
|
| 256 |
+
|
| 257 |
+
def run_portfolio_arena(
|
| 258 |
+
cfg: PortfolioLabConfig,
|
| 259 |
+
strategy_keys: Optional[List[str]] = None,
|
| 260 |
+
n_permutations: int = 80,
|
| 261 |
+
progress: ProgressFn = None,
|
| 262 |
+
) -> tuple[pd.DataFrame, Panel, Dict[str, PortfolioResult]]:
|
| 263 |
+
"""Race every cross-sectional strategy on one universe, ranked by evidence."""
|
| 264 |
+
panel = load_panel(cfg.symbols, cfg.start, cfg.end, cfg.interval, cfg.source)
|
| 265 |
+
ppy = infer_periods_per_year(panel.index)
|
| 266 |
+
schedule = rebalance_schedule(panel.index, cfg.rebalance)
|
| 267 |
+
costs = cfg.costs()
|
| 268 |
+
|
| 269 |
+
keys = strategy_keys or [s.key for s in list_xs_strategies()]
|
| 270 |
+
factors = build_style_factors(panel)
|
| 271 |
+
rows, books = [], {}
|
| 272 |
+
|
| 273 |
+
for i, key in enumerate(keys):
|
| 274 |
+
strategy = get_xs_strategy(key)
|
| 275 |
+
weights = strategy.generate(panel, strategy.defaults())
|
| 276 |
+
result = run_portfolio_backtest(
|
| 277 |
+
panel, weights, costs=costs, lag=cfg.lag, gross_leverage=cfg.gross_leverage,
|
| 278 |
+
max_weight=cfg.max_weight, allow_short=cfg.allow_short,
|
| 279 |
+
initial_capital=cfg.capital, periods_per_year=ppy, rebalance_on=schedule,
|
| 280 |
+
)
|
| 281 |
+
books[key] = result
|
| 282 |
+
|
| 283 |
+
p_value = float("nan")
|
| 284 |
+
if n_permutations > 0:
|
| 285 |
+
p_value = cross_sectional_permutation_test(
|
| 286 |
+
panel, weights, n_permutations=n_permutations, lag=cfg.lag,
|
| 287 |
+
gross_leverage=cfg.gross_leverage, max_weight=cfg.max_weight,
|
| 288 |
+
allow_short=cfg.allow_short, rebalance_on=schedule, seed=cfg.seed,
|
| 289 |
+
observed=result.sharpe,
|
| 290 |
+
).p_value
|
| 291 |
+
|
| 292 |
+
dsr = deflated_sharpe_ratio(
|
| 293 |
+
result.returns.to_numpy(dtype=float), sharpe_annual=result.sharpe,
|
| 294 |
+
periods_per_year=ppy, n_trials=len(strategy.grid(limit=cfg.grid_limit)),
|
| 295 |
+
)
|
| 296 |
+
attr = factor_attribution(result.returns, factors, ppy)
|
| 297 |
+
|
| 298 |
+
rows.append({
|
| 299 |
+
"Strategy": strategy.name,
|
| 300 |
+
"key": key,
|
| 301 |
+
"Family": strategy.family,
|
| 302 |
+
"Return": result.metrics.get("total_return", 0.0),
|
| 303 |
+
"CAGR": result.metrics.get("cagr", 0.0),
|
| 304 |
+
"Sharpe": result.sharpe,
|
| 305 |
+
"MaxDD": result.metrics.get("max_drawdown", 0.0),
|
| 306 |
+
"Turnover": result.metrics.get("turnover_ann", 0.0),
|
| 307 |
+
"p-value": p_value,
|
| 308 |
+
"DSR": dsr["dsr"],
|
| 309 |
+
"Alpha t": attr.get("alpha_t_stat", float("nan")) if attr.get("available") else float("nan"),
|
| 310 |
+
})
|
| 311 |
+
if progress is not None:
|
| 312 |
+
progress((i + 1) / len(keys), f"{strategy.name} ({i + 1}/{len(keys)})")
|
| 313 |
+
|
| 314 |
+
table = pd.DataFrame(rows)
|
| 315 |
+
if not table.empty:
|
| 316 |
+
table["Evidence"] = (1.0 - table["p-value"].fillna(0.5)) * table["DSR"]
|
| 317 |
+
table = table.sort_values("Evidence", ascending=False).reset_index(drop=True)
|
| 318 |
+
table.insert(0, "#", table.index + 1)
|
| 319 |
+
return table, panel, books
|
algotrader/strategies.py
ADDED
|
@@ -0,0 +1,345 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""The strategy zoo.
|
| 2 |
+
|
| 3 |
+
Every strategy is a pure function ``(df, **params) -> target exposure series``
|
| 4 |
+
in ``[-1, 1]``, causal by construction. They never see costs, capital or
|
| 5 |
+
execution — that is the engine's job — which is what lets the same function be
|
| 6 |
+
re-run thousands of times inside the permutation and PBO machinery.
|
| 7 |
+
"""
|
| 8 |
+
|
| 9 |
+
from __future__ import annotations
|
| 10 |
+
|
| 11 |
+
import itertools
|
| 12 |
+
from dataclasses import dataclass
|
| 13 |
+
from typing import Callable, Dict, Iterable, List, Sequence
|
| 14 |
+
|
| 15 |
+
import numpy as np
|
| 16 |
+
import pandas as pd
|
| 17 |
+
|
| 18 |
+
from . import indicators as ind
|
| 19 |
+
|
| 20 |
+
__all__ = ["Strategy", "ParamSpec", "REGISTRY", "get_strategy", "list_strategies"]
|
| 21 |
+
|
| 22 |
+
|
| 23 |
+
@dataclass(frozen=True)
|
| 24 |
+
class ParamSpec:
|
| 25 |
+
name: str
|
| 26 |
+
label: str
|
| 27 |
+
default: float
|
| 28 |
+
grid: Sequence[float]
|
| 29 |
+
kind: str = "int"
|
| 30 |
+
minimum: float | None = None
|
| 31 |
+
maximum: float | None = None
|
| 32 |
+
step: float | None = None
|
| 33 |
+
|
| 34 |
+
def cast(self, value):
|
| 35 |
+
return int(value) if self.kind == "int" else float(value)
|
| 36 |
+
|
| 37 |
+
|
| 38 |
+
@dataclass(frozen=True)
|
| 39 |
+
class Strategy:
|
| 40 |
+
key: str
|
| 41 |
+
name: str
|
| 42 |
+
family: str
|
| 43 |
+
description: str
|
| 44 |
+
fn: Callable[..., pd.Series]
|
| 45 |
+
params: tuple[ParamSpec, ...] = ()
|
| 46 |
+
|
| 47 |
+
def defaults(self) -> Dict[str, float]:
|
| 48 |
+
return {p.name: p.cast(p.default) for p in self.params}
|
| 49 |
+
|
| 50 |
+
def clean(self, params: Dict[str, float] | None) -> Dict[str, float]:
|
| 51 |
+
"""Fill in missing params and coerce types, ignoring unknown keys."""
|
| 52 |
+
merged = self.defaults()
|
| 53 |
+
for spec in self.params:
|
| 54 |
+
if params and spec.name in params and params[spec.name] is not None:
|
| 55 |
+
merged[spec.name] = spec.cast(params[spec.name])
|
| 56 |
+
return merged
|
| 57 |
+
|
| 58 |
+
def generate(self, df: pd.DataFrame, params: Dict[str, float] | None = None) -> pd.Series:
|
| 59 |
+
target = self.fn(df, **self.clean(params))
|
| 60 |
+
return target.reindex(df.index).astype(float).fillna(0.0).clip(-1.0, 1.0)
|
| 61 |
+
|
| 62 |
+
def grid(self, limit: int | None = None) -> List[Dict[str, float]]:
|
| 63 |
+
"""Cartesian product of the per-parameter grids (the 'trials' a
|
| 64 |
+
researcher would realistically run before picking a winner)."""
|
| 65 |
+
if not self.params:
|
| 66 |
+
return [{}]
|
| 67 |
+
names = [p.name for p in self.params]
|
| 68 |
+
combos = [
|
| 69 |
+
dict(zip(names, values))
|
| 70 |
+
for values in itertools.product(*[p.grid for p in self.params])
|
| 71 |
+
]
|
| 72 |
+
combos = [c for c in combos if self._valid(c)]
|
| 73 |
+
if limit is not None and len(combos) > limit:
|
| 74 |
+
step = len(combos) / limit
|
| 75 |
+
combos = [combos[int(i * step)] for i in range(limit)]
|
| 76 |
+
return combos
|
| 77 |
+
|
| 78 |
+
def _valid(self, combo: Dict[str, float]) -> bool:
|
| 79 |
+
"""Reject nonsensical combinations (a fast MA slower than the slow one)."""
|
| 80 |
+
if "fast" in combo and "slow" in combo and combo["fast"] >= combo["slow"]:
|
| 81 |
+
return False
|
| 82 |
+
if "lower" in combo and "upper" in combo and combo["lower"] >= combo["upper"]:
|
| 83 |
+
return False
|
| 84 |
+
return True
|
| 85 |
+
|
| 86 |
+
|
| 87 |
+
def _hold_until_flip(raw: pd.Series) -> pd.Series:
|
| 88 |
+
"""Turn sparse entry/exit signals into a continuously held position."""
|
| 89 |
+
return raw.ffill().fillna(0.0)
|
| 90 |
+
|
| 91 |
+
|
| 92 |
+
# --------------------------------------------------------------------------
|
| 93 |
+
# Strategy implementations
|
| 94 |
+
# --------------------------------------------------------------------------
|
| 95 |
+
|
| 96 |
+
def _buy_and_hold(df: pd.DataFrame) -> pd.Series:
|
| 97 |
+
return pd.Series(1.0, index=df.index)
|
| 98 |
+
|
| 99 |
+
|
| 100 |
+
def _sma_cross(df: pd.DataFrame, fast: int = 20, slow: int = 100) -> pd.Series:
|
| 101 |
+
f, s = ind.sma(df["close"], fast), ind.sma(df["close"], slow)
|
| 102 |
+
return pd.Series(np.where(f > s, 1.0, -1.0), index=df.index).where(s.notna())
|
| 103 |
+
|
| 104 |
+
|
| 105 |
+
def _ema_cross(df: pd.DataFrame, fast: int = 12, slow: int = 50) -> pd.Series:
|
| 106 |
+
f, s = ind.ema(df["close"], fast), ind.ema(df["close"], slow)
|
| 107 |
+
return pd.Series(np.where(f > s, 1.0, -1.0), index=df.index).where(s.notna())
|
| 108 |
+
|
| 109 |
+
|
| 110 |
+
def _macd_trend(df: pd.DataFrame, fast: int = 12, slow: int = 26, signal: int = 9) -> pd.Series:
|
| 111 |
+
_, _, hist = ind.macd(df["close"], fast, slow, signal)
|
| 112 |
+
return pd.Series(np.sign(hist), index=df.index).where(hist.notna())
|
| 113 |
+
|
| 114 |
+
|
| 115 |
+
def _rsi_reversion(df: pd.DataFrame, window: int = 14, lower: int = 30, upper: int = 70) -> pd.Series:
|
| 116 |
+
r = ind.rsi(df["close"], window)
|
| 117 |
+
raw = pd.Series(np.nan, index=df.index)
|
| 118 |
+
raw[r < lower] = 1.0
|
| 119 |
+
raw[r > upper] = -1.0
|
| 120 |
+
raw[(r > 45) & (r < 55)] = 0.0 # flatten in the middle of the range
|
| 121 |
+
return _hold_until_flip(raw).where(r.notna())
|
| 122 |
+
|
| 123 |
+
|
| 124 |
+
def _bollinger_reversion(df: pd.DataFrame, window: int = 20, k: float = 2.0) -> pd.Series:
|
| 125 |
+
low, mid, high = ind.bollinger(df["close"], window, k)
|
| 126 |
+
close = df["close"]
|
| 127 |
+
raw = pd.Series(np.nan, index=df.index)
|
| 128 |
+
raw[close < low] = 1.0
|
| 129 |
+
raw[close > high] = -1.0
|
| 130 |
+
raw[(close - mid).abs() < 0.1 * (high - mid)] = 0.0
|
| 131 |
+
return _hold_until_flip(raw).where(mid.notna())
|
| 132 |
+
|
| 133 |
+
|
| 134 |
+
def _donchian_breakout(df: pd.DataFrame, window: int = 20) -> pd.Series:
|
| 135 |
+
low, high = ind.donchian(df, window)
|
| 136 |
+
raw = pd.Series(np.nan, index=df.index)
|
| 137 |
+
raw[df["close"] > high] = 1.0
|
| 138 |
+
raw[df["close"] < low] = -1.0
|
| 139 |
+
return _hold_until_flip(raw).where(high.notna())
|
| 140 |
+
|
| 141 |
+
|
| 142 |
+
def _momentum(df: pd.DataFrame, lookback: int = 60) -> pd.Series:
|
| 143 |
+
return np.sign(ind.roc(df["close"], lookback))
|
| 144 |
+
|
| 145 |
+
|
| 146 |
+
def _vol_target_momentum(
|
| 147 |
+
df: pd.DataFrame, lookback: int = 60, vol_window: int = 20, target_vol: float = 15
|
| 148 |
+
) -> pd.Series:
|
| 149 |
+
"""Momentum sized inversely to recent volatility (targets ``target_vol`` %)."""
|
| 150 |
+
signal = np.sign(ind.roc(df["close"], lookback))
|
| 151 |
+
rv = ind.realised_vol(df["close"].pct_change(), vol_window)
|
| 152 |
+
scale = (target_vol / 100.0) / rv.replace(0.0, np.nan)
|
| 153 |
+
return (signal * scale.clip(upper=1.0)).where(rv.notna())
|
| 154 |
+
|
| 155 |
+
|
| 156 |
+
def _channel_trend(df: pd.DataFrame, window: int = 50, atr_window: int = 14, mult: float = 1.0) -> pd.Series:
|
| 157 |
+
"""Long above an ATR band around the mean, short below it, flat inside."""
|
| 158 |
+
mid = ind.sma(df["close"], window)
|
| 159 |
+
band = ind.atr(df, atr_window) * mult
|
| 160 |
+
raw = pd.Series(np.nan, index=df.index)
|
| 161 |
+
raw[df["close"] > mid + band] = 1.0
|
| 162 |
+
raw[df["close"] < mid - band] = -1.0
|
| 163 |
+
raw[(df["close"] - mid).abs() < 0.25 * band] = 0.0
|
| 164 |
+
return _hold_until_flip(raw).where(mid.notna() & band.notna())
|
| 165 |
+
|
| 166 |
+
|
| 167 |
+
def _coin_flip(df: pd.DataFrame, hold: int = 5, seed: int = 7) -> pd.Series:
|
| 168 |
+
"""A deliberately worthless strategy: the control group.
|
| 169 |
+
|
| 170 |
+
If your clever rule cannot beat this on the validation panel, that is the
|
| 171 |
+
single most useful thing this app can tell you.
|
| 172 |
+
"""
|
| 173 |
+
rng = np.random.default_rng(int(seed))
|
| 174 |
+
n = len(df)
|
| 175 |
+
draws = rng.choice([-1.0, 1.0], size=int(np.ceil(n / max(hold, 1))))
|
| 176 |
+
return pd.Series(np.repeat(draws, max(hold, 1))[:n], index=df.index)
|
| 177 |
+
|
| 178 |
+
|
| 179 |
+
REGISTRY: Dict[str, Strategy] = {}
|
| 180 |
+
|
| 181 |
+
|
| 182 |
+
def _register(strategy: Strategy) -> Strategy:
|
| 183 |
+
REGISTRY[strategy.key] = strategy
|
| 184 |
+
return strategy
|
| 185 |
+
|
| 186 |
+
|
| 187 |
+
_register(
|
| 188 |
+
Strategy(
|
| 189 |
+
key="buy_and_hold",
|
| 190 |
+
name="Buy & Hold",
|
| 191 |
+
family="benchmark",
|
| 192 |
+
description="Own the asset, do nothing. The bar every other strategy has to clear.",
|
| 193 |
+
fn=_buy_and_hold,
|
| 194 |
+
)
|
| 195 |
+
)
|
| 196 |
+
|
| 197 |
+
_register(
|
| 198 |
+
Strategy(
|
| 199 |
+
key="sma_cross",
|
| 200 |
+
name="SMA Crossover",
|
| 201 |
+
family="trend",
|
| 202 |
+
description="Long when the fast simple moving average is above the slow one, short when below.",
|
| 203 |
+
fn=_sma_cross,
|
| 204 |
+
params=(
|
| 205 |
+
ParamSpec("fast", "Fast MA", 20, (5, 10, 20, 30, 50), "int", 2, 100, 1),
|
| 206 |
+
ParamSpec("slow", "Slow MA", 100, (50, 100, 150, 200), "int", 10, 300, 5),
|
| 207 |
+
),
|
| 208 |
+
)
|
| 209 |
+
)
|
| 210 |
+
|
| 211 |
+
_register(
|
| 212 |
+
Strategy(
|
| 213 |
+
key="ema_cross",
|
| 214 |
+
name="EMA Crossover",
|
| 215 |
+
family="trend",
|
| 216 |
+
description="Same idea as the SMA cross but with exponential averages, so it turns faster.",
|
| 217 |
+
fn=_ema_cross,
|
| 218 |
+
params=(
|
| 219 |
+
ParamSpec("fast", "Fast EMA", 12, (5, 8, 12, 21, 34), "int", 2, 100, 1),
|
| 220 |
+
ParamSpec("slow", "Slow EMA", 50, (34, 50, 89, 144, 200), "int", 10, 300, 1),
|
| 221 |
+
),
|
| 222 |
+
)
|
| 223 |
+
)
|
| 224 |
+
|
| 225 |
+
_register(
|
| 226 |
+
Strategy(
|
| 227 |
+
key="macd_trend",
|
| 228 |
+
name="MACD Trend",
|
| 229 |
+
family="trend",
|
| 230 |
+
description="Follow the sign of the MACD histogram.",
|
| 231 |
+
fn=_macd_trend,
|
| 232 |
+
params=(
|
| 233 |
+
ParamSpec("fast", "Fast", 12, (8, 12, 16), "int", 2, 60, 1),
|
| 234 |
+
ParamSpec("slow", "Slow", 26, (21, 26, 34, 50), "int", 10, 200, 1),
|
| 235 |
+
ParamSpec("signal", "Signal", 9, (5, 9, 13), "int", 2, 50, 1),
|
| 236 |
+
),
|
| 237 |
+
)
|
| 238 |
+
)
|
| 239 |
+
|
| 240 |
+
_register(
|
| 241 |
+
Strategy(
|
| 242 |
+
key="rsi_reversion",
|
| 243 |
+
name="RSI Mean Reversion",
|
| 244 |
+
family="mean-reversion",
|
| 245 |
+
description="Buy oversold, sell overbought, flatten in the middle of the range.",
|
| 246 |
+
fn=_rsi_reversion,
|
| 247 |
+
params=(
|
| 248 |
+
ParamSpec("window", "RSI window", 14, (7, 14, 21), "int", 2, 60, 1),
|
| 249 |
+
ParamSpec("lower", "Oversold", 30, (20, 25, 30, 35), "int", 5, 49, 1),
|
| 250 |
+
ParamSpec("upper", "Overbought", 70, (65, 70, 75, 80), "int", 51, 95, 1),
|
| 251 |
+
),
|
| 252 |
+
)
|
| 253 |
+
)
|
| 254 |
+
|
| 255 |
+
_register(
|
| 256 |
+
Strategy(
|
| 257 |
+
key="bollinger_reversion",
|
| 258 |
+
name="Bollinger Reversion",
|
| 259 |
+
family="mean-reversion",
|
| 260 |
+
description="Fade moves outside the Bollinger bands and exit back at the middle band.",
|
| 261 |
+
fn=_bollinger_reversion,
|
| 262 |
+
params=(
|
| 263 |
+
ParamSpec("window", "Window", 20, (10, 20, 30, 50), "int", 5, 120, 1),
|
| 264 |
+
ParamSpec("k", "Band width (σ)", 2.0, (1.5, 2.0, 2.5, 3.0), "float", 0.5, 4.0, 0.1),
|
| 265 |
+
),
|
| 266 |
+
)
|
| 267 |
+
)
|
| 268 |
+
|
| 269 |
+
_register(
|
| 270 |
+
Strategy(
|
| 271 |
+
key="donchian_breakout",
|
| 272 |
+
name="Donchian Breakout",
|
| 273 |
+
family="breakout",
|
| 274 |
+
description="The classic turtle rule: buy new highs, sell new lows.",
|
| 275 |
+
fn=_donchian_breakout,
|
| 276 |
+
params=(ParamSpec("window", "Channel", 20, (10, 20, 40, 55, 100), "int", 5, 250, 1),),
|
| 277 |
+
)
|
| 278 |
+
)
|
| 279 |
+
|
| 280 |
+
_register(
|
| 281 |
+
Strategy(
|
| 282 |
+
key="momentum",
|
| 283 |
+
name="Time-Series Momentum",
|
| 284 |
+
family="momentum",
|
| 285 |
+
description="Hold long if the asset is up over the lookback, short if it is down.",
|
| 286 |
+
fn=_momentum,
|
| 287 |
+
params=(ParamSpec("lookback", "Lookback", 60, (5, 10, 20, 60, 120, 250), "int", 2, 500, 1),),
|
| 288 |
+
)
|
| 289 |
+
)
|
| 290 |
+
|
| 291 |
+
_register(
|
| 292 |
+
Strategy(
|
| 293 |
+
key="vol_target_momentum",
|
| 294 |
+
name="Vol-Targeted Momentum",
|
| 295 |
+
family="momentum",
|
| 296 |
+
description="Momentum sized down when markets get volatile, so risk stays roughly constant.",
|
| 297 |
+
fn=_vol_target_momentum,
|
| 298 |
+
params=(
|
| 299 |
+
ParamSpec("lookback", "Lookback", 60, (20, 60, 120, 250), "int", 5, 500, 1),
|
| 300 |
+
ParamSpec("vol_window", "Vol window", 20, (10, 20, 60), "int", 5, 120, 1),
|
| 301 |
+
ParamSpec("target_vol", "Target vol %", 15, (10, 15, 20), "float", 2, 60, 1),
|
| 302 |
+
),
|
| 303 |
+
)
|
| 304 |
+
)
|
| 305 |
+
|
| 306 |
+
_register(
|
| 307 |
+
Strategy(
|
| 308 |
+
key="channel_trend",
|
| 309 |
+
name="ATR Channel Trend",
|
| 310 |
+
family="trend",
|
| 311 |
+
description="Trade with the trend only once price clears an ATR band around its mean.",
|
| 312 |
+
fn=_channel_trend,
|
| 313 |
+
params=(
|
| 314 |
+
ParamSpec("window", "Mean window", 50, (20, 50, 100, 200), "int", 5, 300, 1),
|
| 315 |
+
ParamSpec("atr_window", "ATR window", 14, (7, 14, 28), "int", 2, 60, 1),
|
| 316 |
+
ParamSpec("mult", "ATR multiple", 1.0, (0.5, 1.0, 1.5, 2.0), "float", 0.1, 5.0, 0.1),
|
| 317 |
+
),
|
| 318 |
+
)
|
| 319 |
+
)
|
| 320 |
+
|
| 321 |
+
_register(
|
| 322 |
+
Strategy(
|
| 323 |
+
key="coin_flip",
|
| 324 |
+
name="Coin Flip (control)",
|
| 325 |
+
family="control",
|
| 326 |
+
description="Random positions. The control group — anything that cannot beat this is noise.",
|
| 327 |
+
fn=_coin_flip,
|
| 328 |
+
params=(
|
| 329 |
+
ParamSpec("hold", "Bars per flip", 5, (1, 5, 10, 20), "int", 1, 60, 1),
|
| 330 |
+
ParamSpec("seed", "Seed", 7, (1, 7, 42, 123), "int", 0, 9999, 1),
|
| 331 |
+
),
|
| 332 |
+
)
|
| 333 |
+
)
|
| 334 |
+
|
| 335 |
+
|
| 336 |
+
def get_strategy(key: str) -> Strategy:
|
| 337 |
+
try:
|
| 338 |
+
return REGISTRY[key]
|
| 339 |
+
except KeyError:
|
| 340 |
+
raise KeyError(f"Unknown strategy '{key}'. Available: {', '.join(sorted(REGISTRY))}") from None
|
| 341 |
+
|
| 342 |
+
|
| 343 |
+
def list_strategies(exclude: Iterable[str] = ()) -> List[Strategy]:
|
| 344 |
+
skip = set(exclude)
|
| 345 |
+
return [s for k, s in REGISTRY.items() if k not in skip]
|
algotrader/types.py
ADDED
|
@@ -0,0 +1,100 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Core data types shared across the algotrader 2.0 stack.
|
| 2 |
+
|
| 3 |
+
Everything downstream (engine, validation, UI) speaks these types, so they are
|
| 4 |
+
deliberately small, immutable-ish and free of framework dependencies.
|
| 5 |
+
"""
|
| 6 |
+
|
| 7 |
+
from __future__ import annotations
|
| 8 |
+
|
| 9 |
+
from dataclasses import dataclass, field
|
| 10 |
+
from typing import Any, Dict, Optional
|
| 11 |
+
|
| 12 |
+
import pandas as pd
|
| 13 |
+
|
| 14 |
+
OHLCV_COLUMNS = ("open", "high", "low", "close", "volume")
|
| 15 |
+
|
| 16 |
+
|
| 17 |
+
@dataclass(frozen=True)
|
| 18 |
+
class MarketData:
|
| 19 |
+
"""A validated OHLCV series plus provenance.
|
| 20 |
+
|
| 21 |
+
Provenance matters here: the app is about honesty, so the UI always tells
|
| 22 |
+
the user whether they are looking at real prices or a simulation.
|
| 23 |
+
"""
|
| 24 |
+
|
| 25 |
+
symbol: str
|
| 26 |
+
df: pd.DataFrame
|
| 27 |
+
source: str # "yfinance" | "bundled" | "synthetic"
|
| 28 |
+
interval: str = "1d"
|
| 29 |
+
note: str = ""
|
| 30 |
+
|
| 31 |
+
@property
|
| 32 |
+
def is_real(self) -> bool:
|
| 33 |
+
return self.source in ("yfinance", "bundled")
|
| 34 |
+
|
| 35 |
+
@property
|
| 36 |
+
def start(self) -> pd.Timestamp:
|
| 37 |
+
return self.df.index[0]
|
| 38 |
+
|
| 39 |
+
@property
|
| 40 |
+
def end(self) -> pd.Timestamp:
|
| 41 |
+
return self.df.index[-1]
|
| 42 |
+
|
| 43 |
+
def __len__(self) -> int: # pragma: no cover - trivial
|
| 44 |
+
return len(self.df)
|
| 45 |
+
|
| 46 |
+
|
| 47 |
+
@dataclass(frozen=True)
|
| 48 |
+
class CostModel:
|
| 49 |
+
"""Round-trip friction. All values are one-way, in basis points."""
|
| 50 |
+
|
| 51 |
+
commission_bps: float = 1.0
|
| 52 |
+
slippage_bps: float = 2.0
|
| 53 |
+
short_borrow_bps: float = 50.0 # annualised, charged on short exposure
|
| 54 |
+
|
| 55 |
+
@property
|
| 56 |
+
def one_way_bps(self) -> float:
|
| 57 |
+
return self.commission_bps + self.slippage_bps
|
| 58 |
+
|
| 59 |
+
|
| 60 |
+
@dataclass
|
| 61 |
+
class BacktestResult:
|
| 62 |
+
"""Output of a single backtest run."""
|
| 63 |
+
|
| 64 |
+
equity: pd.Series
|
| 65 |
+
returns: pd.Series # net of costs
|
| 66 |
+
gross_returns: pd.Series
|
| 67 |
+
position: pd.Series # exposure actually held during each bar
|
| 68 |
+
target: pd.Series # exposure requested by the strategy
|
| 69 |
+
costs: pd.Series
|
| 70 |
+
benchmark_equity: pd.Series
|
| 71 |
+
metrics: Dict[str, float] = field(default_factory=dict)
|
| 72 |
+
benchmark_metrics: Dict[str, float] = field(default_factory=dict)
|
| 73 |
+
meta: Dict[str, Any] = field(default_factory=dict)
|
| 74 |
+
|
| 75 |
+
@property
|
| 76 |
+
def sharpe(self) -> float:
|
| 77 |
+
return float(self.metrics.get("sharpe", 0.0))
|
| 78 |
+
|
| 79 |
+
@property
|
| 80 |
+
def n_trades(self) -> int:
|
| 81 |
+
return int(self.metrics.get("n_trades", 0))
|
| 82 |
+
|
| 83 |
+
|
| 84 |
+
@dataclass
|
| 85 |
+
class ValidationReport:
|
| 86 |
+
"""Everything we know about how much of a backtest is luck."""
|
| 87 |
+
|
| 88 |
+
permutation_p_value: Optional[float] = None
|
| 89 |
+
permutation_null: Optional[Any] = None # np.ndarray of null Sharpes
|
| 90 |
+
deflated_sharpe: Optional[float] = None
|
| 91 |
+
probabilistic_sharpe: Optional[float] = None
|
| 92 |
+
min_track_record_years: Optional[float] = None
|
| 93 |
+
n_trials: int = 1
|
| 94 |
+
pbo: Optional[float] = None
|
| 95 |
+
pbo_detail: Dict[str, Any] = field(default_factory=dict)
|
| 96 |
+
walkforward: Dict[str, Any] = field(default_factory=dict)
|
| 97 |
+
reality_score: float = 0.0
|
| 98 |
+
grade: str = "?"
|
| 99 |
+
verdict: str = ""
|
| 100 |
+
flags: list = field(default_factory=list)
|
algotrader/validation/__init__.py
ADDED
|
@@ -0,0 +1,15 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Statistical tests that ask "is this edge real, or did we just get lucky?"."""
|
| 2 |
+
|
| 3 |
+
from .deflated_sharpe import deflated_sharpe_ratio, min_track_record_length, probabilistic_sharpe_ratio
|
| 4 |
+
from .pbo import probability_of_backtest_overfitting
|
| 5 |
+
from .permutation import permutation_test
|
| 6 |
+
from .walkforward import walk_forward
|
| 7 |
+
|
| 8 |
+
__all__ = [
|
| 9 |
+
"deflated_sharpe_ratio",
|
| 10 |
+
"probabilistic_sharpe_ratio",
|
| 11 |
+
"min_track_record_length",
|
| 12 |
+
"probability_of_backtest_overfitting",
|
| 13 |
+
"permutation_test",
|
| 14 |
+
"walk_forward",
|
| 15 |
+
]
|
algotrader/validation/cross_permutation.py
ADDED
|
@@ -0,0 +1,146 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""The cross-sectional null.
|
| 2 |
+
|
| 3 |
+
For a timing rule, shuffling the price path is the right null. For a rule that
|
| 4 |
+
*ranks names*, it is the wrong test entirely: shuffling time destroys the
|
| 5 |
+
market's whole correlation structure, and the resulting null is so weak that
|
| 6 |
+
almost any long-short book clears it.
|
| 7 |
+
|
| 8 |
+
The question a cross-sectional strategy has to answer is narrower. Not "does
|
| 9 |
+
this market have structure?" but: **given these dates, these assets and this
|
| 10 |
+
book's shape, does the strategy put its weight on the right names?**
|
| 11 |
+
|
| 12 |
+
So we permute the *weights across assets within each date*. Every calendar
|
| 13 |
+
effect survives. Every correlation between names survives. The gross and net
|
| 14 |
+
exposure of the book on each date survives exactly. The one thing destroyed is
|
| 15 |
+
the link between the strategy's choice and the asset it chose.
|
| 16 |
+
|
| 17 |
+
A momentum book that beats this null is picking names. One that does not was
|
| 18 |
+
being paid for its market exposure, its sector tilt, or the calendar -- all of
|
| 19 |
+
which are available far more cheaply.
|
| 20 |
+
"""
|
| 21 |
+
|
| 22 |
+
from __future__ import annotations
|
| 23 |
+
|
| 24 |
+
from dataclasses import dataclass
|
| 25 |
+
from typing import Callable, Optional
|
| 26 |
+
|
| 27 |
+
import numpy as np
|
| 28 |
+
import pandas as pd
|
| 29 |
+
|
| 30 |
+
from ..panel import Panel
|
| 31 |
+
from ..portfolio import run_portfolio_backtest
|
| 32 |
+
from ..types import CostModel
|
| 33 |
+
|
| 34 |
+
__all__ = ["cross_sectional_permutation_test", "permute_within_dates", "CrossPermutationResult"]
|
| 35 |
+
|
| 36 |
+
|
| 37 |
+
@dataclass
|
| 38 |
+
class CrossPermutationResult:
|
| 39 |
+
observed: float
|
| 40 |
+
null: np.ndarray
|
| 41 |
+
p_value: float
|
| 42 |
+
n_permutations: int
|
| 43 |
+
|
| 44 |
+
@property
|
| 45 |
+
def null_mean(self) -> float:
|
| 46 |
+
return float(np.mean(self.null)) if self.null.size else 0.0
|
| 47 |
+
|
| 48 |
+
@property
|
| 49 |
+
def percentile(self) -> float:
|
| 50 |
+
if not self.null.size:
|
| 51 |
+
return 50.0
|
| 52 |
+
return float((self.null < self.observed).mean() * 100.0)
|
| 53 |
+
|
| 54 |
+
|
| 55 |
+
def permute_within_dates(
|
| 56 |
+
weights: pd.DataFrame,
|
| 57 |
+
investable: pd.DataFrame,
|
| 58 |
+
rng: np.random.Generator,
|
| 59 |
+
) -> pd.DataFrame:
|
| 60 |
+
"""Reassign each date's weights among that date's investable assets.
|
| 61 |
+
|
| 62 |
+
The multiset of weights on every row is preserved exactly -- so gross
|
| 63 |
+
exposure, net exposure, leg sizes and position counts are all identical to
|
| 64 |
+
the real book -- but which asset receives which weight is randomised.
|
| 65 |
+
|
| 66 |
+
Vectorised across all dates at once: sorting each row puts the investable
|
| 67 |
+
weights first and pushes non-investable slots to NaN, then a random rank per
|
| 68 |
+
investable slot picks each weight exactly once.
|
| 69 |
+
"""
|
| 70 |
+
values = weights.to_numpy(dtype=float, copy=True)
|
| 71 |
+
mask = investable.to_numpy(dtype=bool)
|
| 72 |
+
|
| 73 |
+
masked = np.where(mask, values, np.nan)
|
| 74 |
+
# NaNs sort last, so the first k entries of each row are that row's real weights.
|
| 75 |
+
ordered = np.sort(masked, axis=1)
|
| 76 |
+
|
| 77 |
+
noise = np.where(mask, rng.random(values.shape), np.inf)
|
| 78 |
+
# Double argsort turns random values into ranks 0..k-1 for investable slots,
|
| 79 |
+
# and k..n-1 for the rest, which then index into the NaN tail.
|
| 80 |
+
random_rank = np.argsort(np.argsort(noise, axis=1), axis=1)
|
| 81 |
+
|
| 82 |
+
shuffled = np.take_along_axis(ordered, random_rank, axis=1)
|
| 83 |
+
return pd.DataFrame(
|
| 84 |
+
np.nan_to_num(shuffled, nan=0.0), index=weights.index, columns=weights.columns
|
| 85 |
+
)
|
| 86 |
+
|
| 87 |
+
|
| 88 |
+
def cross_sectional_permutation_test(
|
| 89 |
+
panel: Panel,
|
| 90 |
+
weights: pd.DataFrame,
|
| 91 |
+
n_permutations: int = 200,
|
| 92 |
+
costs: Optional[CostModel] = None,
|
| 93 |
+
lag: int = 1,
|
| 94 |
+
gross_leverage: float = 1.0,
|
| 95 |
+
max_weight: Optional[float] = None,
|
| 96 |
+
allow_short: bool = True,
|
| 97 |
+
rebalance_on: Optional[pd.Series] = None,
|
| 98 |
+
seed: int = 0,
|
| 99 |
+
observed: Optional[float] = None,
|
| 100 |
+
neutralise_costs: bool = True,
|
| 101 |
+
progress: Optional[Callable[[float, str], None]] = None,
|
| 102 |
+
) -> CrossPermutationResult:
|
| 103 |
+
"""Test whether a book's Sharpe survives randomising which names it picked.
|
| 104 |
+
|
| 105 |
+
``neutralise_costs`` defaults to True, and it matters more than it looks.
|
| 106 |
+
A real momentum book holds many of the same names from one rebalance to the
|
| 107 |
+
next, so it churns slowly. A shuffled book reassigns names at random every
|
| 108 |
+
date, so it churns furiously and pays for it. Charging costs would penalise
|
| 109 |
+
the null for turnover the strategy never had, and the strategy would look
|
| 110 |
+
good by comparison for reasons that have nothing to do with skill.
|
| 111 |
+
|
| 112 |
+
So this test asks only "did it pick the right names?" and leaves "can you
|
| 113 |
+
afford to trade it?" to the cost stress test, which measures that directly.
|
| 114 |
+
"""
|
| 115 |
+
costs = CostModel(0.0, 0.0, 0.0) if neutralise_costs else (costs or CostModel())
|
| 116 |
+
rng = np.random.default_rng(seed)
|
| 117 |
+
investable = panel.close.notna()
|
| 118 |
+
|
| 119 |
+
def sharpe_of(w: pd.DataFrame) -> float:
|
| 120 |
+
return run_portfolio_backtest(
|
| 121 |
+
panel,
|
| 122 |
+
w,
|
| 123 |
+
costs=costs,
|
| 124 |
+
lag=lag,
|
| 125 |
+
gross_leverage=gross_leverage,
|
| 126 |
+
max_weight=max_weight,
|
| 127 |
+
allow_short=allow_short,
|
| 128 |
+
rebalance_on=rebalance_on,
|
| 129 |
+
).sharpe
|
| 130 |
+
|
| 131 |
+
if observed is None:
|
| 132 |
+
observed = sharpe_of(weights)
|
| 133 |
+
|
| 134 |
+
null = np.empty(n_permutations, dtype=float)
|
| 135 |
+
for i in range(n_permutations):
|
| 136 |
+
null[i] = sharpe_of(permute_within_dates(weights, investable, rng))
|
| 137 |
+
if progress is not None and (i % 10 == 0 or i == n_permutations - 1):
|
| 138 |
+
progress((i + 1) / n_permutations, f"Cross-sectional shuffle {i + 1}/{n_permutations}")
|
| 139 |
+
|
| 140 |
+
p_value = float((1 + np.sum(null >= observed)) / (n_permutations + 1))
|
| 141 |
+
return CrossPermutationResult(
|
| 142 |
+
observed=float(observed),
|
| 143 |
+
null=null,
|
| 144 |
+
p_value=p_value,
|
| 145 |
+
n_permutations=n_permutations,
|
| 146 |
+
)
|
algotrader/validation/deflated_sharpe.py
ADDED
|
@@ -0,0 +1,145 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Probabilistic and Deflated Sharpe Ratios.
|
| 2 |
+
|
| 3 |
+
Bailey & López de Prado (2014), "The Deflated Sharpe Ratio: Correcting for
|
| 4 |
+
Selection Bias, Backtest Overfitting and Non-Normality".
|
| 5 |
+
|
| 6 |
+
The intuition: if you try 200 strategy variants, the best one will show a
|
| 7 |
+
handsome Sharpe *even when none of them has any edge*. The Deflated Sharpe
|
| 8 |
+
Ratio asks whether the winner beats what the luckiest of 200 coin-flippers
|
| 9 |
+
would have produced, and it charges extra for fat tails and negative skew --
|
| 10 |
+
exactly the return shapes that make naive Sharpe ratios flatter.
|
| 11 |
+
"""
|
| 12 |
+
|
| 13 |
+
from __future__ import annotations
|
| 14 |
+
|
| 15 |
+
import numpy as np
|
| 16 |
+
from scipy import stats
|
| 17 |
+
|
| 18 |
+
__all__ = [
|
| 19 |
+
"probabilistic_sharpe_ratio",
|
| 20 |
+
"expected_max_sharpe",
|
| 21 |
+
"deflated_sharpe_ratio",
|
| 22 |
+
"min_track_record_length",
|
| 23 |
+
]
|
| 24 |
+
|
| 25 |
+
_EULER = 0.5772156649015329
|
| 26 |
+
|
| 27 |
+
|
| 28 |
+
def _moments(returns: np.ndarray) -> tuple[float, float]:
|
| 29 |
+
"""Sample skew and *non-excess* kurtosis, as the PSR formula expects."""
|
| 30 |
+
arr = np.asarray(returns, dtype=float)
|
| 31 |
+
arr = arr[np.isfinite(arr)]
|
| 32 |
+
if arr.size < 4:
|
| 33 |
+
return 0.0, 3.0
|
| 34 |
+
return float(stats.skew(arr, bias=False)), float(stats.kurtosis(arr, bias=False) + 3.0)
|
| 35 |
+
|
| 36 |
+
|
| 37 |
+
def probabilistic_sharpe_ratio(
|
| 38 |
+
sharpe: float,
|
| 39 |
+
n_obs: int,
|
| 40 |
+
skew: float = 0.0,
|
| 41 |
+
kurtosis: float = 3.0,
|
| 42 |
+
benchmark: float = 0.0,
|
| 43 |
+
) -> float:
|
| 44 |
+
"""P(true Sharpe > ``benchmark``) given the observed Sharpe and its shape.
|
| 45 |
+
|
| 46 |
+
``sharpe`` and ``benchmark`` are per-observation (i.e. *not* annualised).
|
| 47 |
+
"""
|
| 48 |
+
if n_obs < 3:
|
| 49 |
+
return 0.5
|
| 50 |
+
denom = 1.0 - skew * sharpe + ((kurtosis - 1.0) / 4.0) * sharpe**2
|
| 51 |
+
if denom <= 0:
|
| 52 |
+
return 0.5
|
| 53 |
+
z = (sharpe - benchmark) * np.sqrt(n_obs - 1) / np.sqrt(denom)
|
| 54 |
+
return float(stats.norm.cdf(z))
|
| 55 |
+
|
| 56 |
+
|
| 57 |
+
def expected_max_sharpe(n_trials: int, variance_of_trials: float) -> float:
|
| 58 |
+
"""Expected maximum Sharpe across ``n_trials`` *skill-free* strategies.
|
| 59 |
+
|
| 60 |
+
This is the bar the winner has to clear to be interesting. It grows with
|
| 61 |
+
the number of things you tried — which is why "I found a strategy with
|
| 62 |
+
Sharpe 2" means nothing until you say how many you looked at.
|
| 63 |
+
"""
|
| 64 |
+
n = max(int(n_trials), 1)
|
| 65 |
+
if n == 1 or variance_of_trials <= 0:
|
| 66 |
+
return 0.0
|
| 67 |
+
sd = np.sqrt(variance_of_trials)
|
| 68 |
+
# Bailey & López de Prado's Gumbel-based approximation.
|
| 69 |
+
q1 = stats.norm.ppf(1.0 - 1.0 / n)
|
| 70 |
+
q2 = stats.norm.ppf(1.0 - 1.0 / (n * np.e))
|
| 71 |
+
return float(sd * ((1.0 - _EULER) * q1 + _EULER * q2))
|
| 72 |
+
|
| 73 |
+
|
| 74 |
+
def deflated_sharpe_ratio(
|
| 75 |
+
returns,
|
| 76 |
+
sharpe_annual: float,
|
| 77 |
+
periods_per_year: int,
|
| 78 |
+
n_trials: int,
|
| 79 |
+
trial_sharpes=None,
|
| 80 |
+
variance_of_trials: float | None = None,
|
| 81 |
+
) -> dict:
|
| 82 |
+
"""Deflate an annualised Sharpe for selection bias and non-normality.
|
| 83 |
+
|
| 84 |
+
Returns a dict with the PSR against a zero benchmark, the selection-bias
|
| 85 |
+
threshold, the deflated probability, and the inputs used, so the UI can
|
| 86 |
+
show its working rather than just a number.
|
| 87 |
+
"""
|
| 88 |
+
arr = np.asarray(returns, dtype=float)
|
| 89 |
+
arr = arr[np.isfinite(arr)]
|
| 90 |
+
n_obs = arr.size
|
| 91 |
+
sr_per_period = sharpe_annual / np.sqrt(periods_per_year)
|
| 92 |
+
skew, kurt = _moments(arr)
|
| 93 |
+
|
| 94 |
+
if variance_of_trials is None:
|
| 95 |
+
if trial_sharpes is not None and len(trial_sharpes) > 1:
|
| 96 |
+
trials = np.asarray(trial_sharpes, dtype=float) / np.sqrt(periods_per_year)
|
| 97 |
+
trials = trials[np.isfinite(trials)]
|
| 98 |
+
variance_of_trials = float(np.var(trials, ddof=1)) if trials.size > 1 else 0.0
|
| 99 |
+
else:
|
| 100 |
+
# With no trial cloud to measure, fall back to the asymptotic
|
| 101 |
+
# variance of a skill-free Sharpe estimate.
|
| 102 |
+
variance_of_trials = 1.0 / max(n_obs - 1, 1)
|
| 103 |
+
|
| 104 |
+
threshold = expected_max_sharpe(n_trials, variance_of_trials)
|
| 105 |
+
|
| 106 |
+
psr = probabilistic_sharpe_ratio(sr_per_period, n_obs, skew, kurt, 0.0)
|
| 107 |
+
dsr = probabilistic_sharpe_ratio(sr_per_period, n_obs, skew, kurt, threshold)
|
| 108 |
+
|
| 109 |
+
return {
|
| 110 |
+
"psr": float(psr),
|
| 111 |
+
"dsr": float(dsr),
|
| 112 |
+
"sr_per_period": float(sr_per_period),
|
| 113 |
+
"threshold_sr_per_period": float(threshold),
|
| 114 |
+
"threshold_sr_annual": float(threshold * np.sqrt(periods_per_year)),
|
| 115 |
+
"n_obs": int(n_obs),
|
| 116 |
+
"n_trials": int(n_trials),
|
| 117 |
+
"skew": float(skew),
|
| 118 |
+
"kurtosis": float(kurt),
|
| 119 |
+
"variance_of_trials": float(variance_of_trials),
|
| 120 |
+
}
|
| 121 |
+
|
| 122 |
+
|
| 123 |
+
def min_track_record_length(
|
| 124 |
+
sharpe: float,
|
| 125 |
+
n_obs: int,
|
| 126 |
+
skew: float = 0.0,
|
| 127 |
+
kurtosis: float = 3.0,
|
| 128 |
+
benchmark: float = 0.0,
|
| 129 |
+
confidence: float = 0.95,
|
| 130 |
+
) -> float:
|
| 131 |
+
"""Observations needed before the Sharpe is significant at ``confidence``.
|
| 132 |
+
|
| 133 |
+
Inputs are per-observation. Returns ``inf`` when the edge is too small to
|
| 134 |
+
ever clear the bar.
|
| 135 |
+
"""
|
| 136 |
+
if sharpe <= benchmark:
|
| 137 |
+
return float("inf")
|
| 138 |
+
z = stats.norm.ppf(confidence)
|
| 139 |
+
denom = (sharpe - benchmark) ** 2
|
| 140 |
+
if denom <= 0:
|
| 141 |
+
return float("inf")
|
| 142 |
+
numer = 1.0 - skew * sharpe + ((kurtosis - 1.0) / 4.0) * sharpe**2
|
| 143 |
+
if numer <= 0:
|
| 144 |
+
return float("inf")
|
| 145 |
+
return float(1.0 + numer * (z / (sharpe - benchmark)) ** 2)
|
algotrader/validation/pbo.py
ADDED
|
@@ -0,0 +1,120 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Probability of Backtest Overfitting via Combinatorially Symmetric CV.
|
| 2 |
+
|
| 3 |
+
Bailey, Borwein, López de Prado & Zhu (2016), "The Probability of Backtest
|
| 4 |
+
Overfitting".
|
| 5 |
+
|
| 6 |
+
Take the N parameter variants you tried, cut the timeline into S chunks, and
|
| 7 |
+
for every way of splitting those chunks half-and-half: pick the variant that
|
| 8 |
+
won in-sample, then look up where it ranked out-of-sample. If your selection
|
| 9 |
+
process has skill, the in-sample winner should keep winning. If it is fitting
|
| 10 |
+
noise, the winner lands in the bottom half about as often as not — and PBO
|
| 11 |
+
approaches 0.5.
|
| 12 |
+
"""
|
| 13 |
+
|
| 14 |
+
from __future__ import annotations
|
| 15 |
+
|
| 16 |
+
import itertools
|
| 17 |
+
from typing import Dict, Sequence
|
| 18 |
+
|
| 19 |
+
import numpy as np
|
| 20 |
+
|
| 21 |
+
__all__ = ["probability_of_backtest_overfitting"]
|
| 22 |
+
|
| 23 |
+
|
| 24 |
+
def _sharpe_columns(matrix: np.ndarray) -> np.ndarray:
|
| 25 |
+
"""Per-column Sharpe (per-period, unannualised — ranks are all we need)."""
|
| 26 |
+
if matrix.shape[0] < 2:
|
| 27 |
+
return np.zeros(matrix.shape[1])
|
| 28 |
+
mean = matrix.mean(axis=0)
|
| 29 |
+
sd = matrix.std(axis=0, ddof=1)
|
| 30 |
+
# Absolute floor, not `> 0`: a flat column's std is float noise, and
|
| 31 |
+
# dividing by it would hand a do-nothing variant an enormous rank.
|
| 32 |
+
with np.errstate(divide="ignore", invalid="ignore"):
|
| 33 |
+
out = np.where(sd > 1e-12, mean / sd, 0.0)
|
| 34 |
+
return np.nan_to_num(out, nan=0.0, posinf=0.0, neginf=0.0)
|
| 35 |
+
|
| 36 |
+
|
| 37 |
+
def probability_of_backtest_overfitting(
|
| 38 |
+
returns_matrix: np.ndarray,
|
| 39 |
+
n_splits: int = 8,
|
| 40 |
+
labels: Sequence[str] | None = None,
|
| 41 |
+
max_combinations: int = 200,
|
| 42 |
+
) -> Dict[str, object]:
|
| 43 |
+
"""Compute PBO for a ``T x N`` matrix of per-period strategy returns.
|
| 44 |
+
|
| 45 |
+
``n_splits`` must be even. Returns PBO, the logit distribution, the
|
| 46 |
+
in-sample/out-of-sample Sharpe pairs for the selected variants, and the
|
| 47 |
+
rate at which the selected variant actually loses money out of sample.
|
| 48 |
+
"""
|
| 49 |
+
matrix = np.asarray(returns_matrix, dtype=float)
|
| 50 |
+
if matrix.ndim != 2:
|
| 51 |
+
raise ValueError("returns_matrix must be 2-D (time x strategy)")
|
| 52 |
+
matrix = np.nan_to_num(matrix, nan=0.0, posinf=0.0, neginf=0.0)
|
| 53 |
+
t_obs, n_strats = matrix.shape
|
| 54 |
+
|
| 55 |
+
if n_strats < 2 or t_obs < 2 * n_splits:
|
| 56 |
+
return {
|
| 57 |
+
"pbo": float("nan"),
|
| 58 |
+
"n_strategies": int(n_strats),
|
| 59 |
+
"n_combinations": 0,
|
| 60 |
+
"logits": np.array([]),
|
| 61 |
+
"is_sharpes": np.array([]),
|
| 62 |
+
"oos_sharpes": np.array([]),
|
| 63 |
+
"prob_oos_loss": float("nan"),
|
| 64 |
+
"performance_degradation": float("nan"),
|
| 65 |
+
"note": "Not enough variants or observations to estimate PBO.",
|
| 66 |
+
}
|
| 67 |
+
|
| 68 |
+
if n_splits % 2:
|
| 69 |
+
n_splits += 1
|
| 70 |
+
chunks = np.array_split(np.arange(t_obs), n_splits)
|
| 71 |
+
|
| 72 |
+
combos = list(itertools.combinations(range(n_splits), n_splits // 2))
|
| 73 |
+
if len(combos) > max_combinations:
|
| 74 |
+
step = len(combos) / max_combinations
|
| 75 |
+
combos = [combos[int(i * step)] for i in range(max_combinations)]
|
| 76 |
+
|
| 77 |
+
logits, is_sr, oos_sr, chosen = [], [], [], []
|
| 78 |
+
for combo in combos:
|
| 79 |
+
is_idx = np.concatenate([chunks[c] for c in combo])
|
| 80 |
+
oos_idx = np.concatenate([chunks[c] for c in range(n_splits) if c not in combo])
|
| 81 |
+
|
| 82 |
+
is_perf = _sharpe_columns(matrix[is_idx])
|
| 83 |
+
oos_perf = _sharpe_columns(matrix[oos_idx])
|
| 84 |
+
|
| 85 |
+
best = int(np.argmax(is_perf))
|
| 86 |
+
chosen.append(best)
|
| 87 |
+
is_sr.append(float(is_perf[best]))
|
| 88 |
+
oos_sr.append(float(oos_perf[best]))
|
| 89 |
+
|
| 90 |
+
# Relative rank of the chosen variant in the OOS ranking, in (0, 1).
|
| 91 |
+
rank = float(np.sum(oos_perf <= oos_perf[best]))
|
| 92 |
+
omega = rank / (n_strats + 1.0)
|
| 93 |
+
omega = min(max(omega, 1e-6), 1.0 - 1e-6)
|
| 94 |
+
logits.append(float(np.log(omega / (1.0 - omega))))
|
| 95 |
+
|
| 96 |
+
logits_arr = np.asarray(logits)
|
| 97 |
+
is_arr, oos_arr = np.asarray(is_sr), np.asarray(oos_sr)
|
| 98 |
+
|
| 99 |
+
# Slope of OOS on IS: negative means better in-sample fits do *worse* live.
|
| 100 |
+
degradation = float("nan")
|
| 101 |
+
if is_arr.size > 2 and np.std(is_arr) > 1e-12:
|
| 102 |
+
degradation = float(np.polyfit(is_arr, oos_arr, 1)[0])
|
| 103 |
+
|
| 104 |
+
counts = np.bincount(chosen, minlength=n_strats)
|
| 105 |
+
most_selected = int(np.argmax(counts))
|
| 106 |
+
|
| 107 |
+
return {
|
| 108 |
+
"pbo": float(np.mean(logits_arr <= 0.0)),
|
| 109 |
+
"n_strategies": int(n_strats),
|
| 110 |
+
"n_combinations": int(len(combos)),
|
| 111 |
+
"logits": logits_arr,
|
| 112 |
+
"is_sharpes": is_arr,
|
| 113 |
+
"oos_sharpes": oos_arr,
|
| 114 |
+
"prob_oos_loss": float(np.mean(oos_arr <= 0.0)),
|
| 115 |
+
"performance_degradation": degradation,
|
| 116 |
+
"most_selected_index": most_selected,
|
| 117 |
+
"most_selected_label": (labels[most_selected] if labels is not None else str(most_selected)),
|
| 118 |
+
"selection_stability": float(counts[most_selected] / max(len(combos), 1)),
|
| 119 |
+
"note": "",
|
| 120 |
+
}
|
algotrader/validation/permutation.py
ADDED
|
@@ -0,0 +1,187 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Monte-Carlo permutation test for trading rules.
|
| 2 |
+
|
| 3 |
+
The question this answers is not "did the strategy make money" but "would a
|
| 4 |
+
rule of this shape have made this much money on a market with no exploitable
|
| 5 |
+
structure?". We destroy the serial dependence in the price path while keeping
|
| 6 |
+
its distribution of moves intact, re-run the *same* strategy on each shuffled
|
| 7 |
+
market, and see where the real result lands in that null distribution.
|
| 8 |
+
|
| 9 |
+
A strategy whose Sharpe sits comfortably inside the null is not a strategy —
|
| 10 |
+
it is a lottery ticket that happened to win.
|
| 11 |
+
"""
|
| 12 |
+
|
| 13 |
+
from __future__ import annotations
|
| 14 |
+
|
| 15 |
+
from dataclasses import dataclass
|
| 16 |
+
from typing import Callable, Optional
|
| 17 |
+
|
| 18 |
+
import numpy as np
|
| 19 |
+
import pandas as pd
|
| 20 |
+
|
| 21 |
+
from ..engine import bars_to_returns, run_backtest
|
| 22 |
+
from ..types import CostModel
|
| 23 |
+
|
| 24 |
+
__all__ = ["permutation_test", "PermutationResult", "permute_bars"]
|
| 25 |
+
|
| 26 |
+
|
| 27 |
+
@dataclass
|
| 28 |
+
class PermutationResult:
|
| 29 |
+
observed: float
|
| 30 |
+
null: np.ndarray
|
| 31 |
+
p_value: float
|
| 32 |
+
method: str
|
| 33 |
+
n_permutations: int
|
| 34 |
+
|
| 35 |
+
@property
|
| 36 |
+
def null_mean(self) -> float:
|
| 37 |
+
return float(np.mean(self.null)) if self.null.size else 0.0
|
| 38 |
+
|
| 39 |
+
@property
|
| 40 |
+
def percentile(self) -> float:
|
| 41 |
+
"""Where the observed Sharpe sits in the null distribution, 0-100."""
|
| 42 |
+
if not self.null.size:
|
| 43 |
+
return 50.0
|
| 44 |
+
return float((self.null < self.observed).mean() * 100.0)
|
| 45 |
+
|
| 46 |
+
|
| 47 |
+
def _decompose(df: pd.DataFrame) -> tuple[np.ndarray, float]:
|
| 48 |
+
"""Split bars into scale-free log moves that can be reshuffled safely."""
|
| 49 |
+
open_ = df["open"].to_numpy(dtype=float)
|
| 50 |
+
high = df["high"].to_numpy(dtype=float)
|
| 51 |
+
low = df["low"].to_numpy(dtype=float)
|
| 52 |
+
close = df["close"].to_numpy(dtype=float)
|
| 53 |
+
volume = df["volume"].to_numpy(dtype=float)
|
| 54 |
+
|
| 55 |
+
gap = np.log(open_[1:] / close[:-1])
|
| 56 |
+
hi = np.log(np.maximum(high[1:], open_[1:]) / open_[1:])
|
| 57 |
+
lo = np.log(np.minimum(low[1:], open_[1:]) / open_[1:])
|
| 58 |
+
body = np.log(close[1:] / open_[1:])
|
| 59 |
+
return np.column_stack([gap, hi, lo, body, volume[1:]]), float(close[0])
|
| 60 |
+
|
| 61 |
+
|
| 62 |
+
def _rebuild(parts: np.ndarray, anchor: float, index: pd.Index, first_row: pd.Series) -> pd.DataFrame:
|
| 63 |
+
gap, hi, lo, body, volume = (parts[:, i] for i in range(5))
|
| 64 |
+
n = parts.shape[0] + 1
|
| 65 |
+
|
| 66 |
+
close = np.empty(n)
|
| 67 |
+
open_ = np.empty(n)
|
| 68 |
+
high = np.empty(n)
|
| 69 |
+
low = np.empty(n)
|
| 70 |
+
vol = np.empty(n)
|
| 71 |
+
|
| 72 |
+
close[0] = anchor
|
| 73 |
+
open_[0] = float(first_row["open"])
|
| 74 |
+
high[0] = float(first_row["high"])
|
| 75 |
+
low[0] = float(first_row["low"])
|
| 76 |
+
vol[0] = float(first_row["volume"])
|
| 77 |
+
|
| 78 |
+
# Cumulative product form: close[i] = close[0] * exp(cumsum(gap + body)).
|
| 79 |
+
close[1:] = anchor * np.exp(np.cumsum(gap + body))
|
| 80 |
+
open_[1:] = close[:-1] * np.exp(gap)
|
| 81 |
+
high[1:] = open_[1:] * np.exp(hi)
|
| 82 |
+
low[1:] = open_[1:] * np.exp(lo)
|
| 83 |
+
vol[1:] = volume
|
| 84 |
+
|
| 85 |
+
return pd.DataFrame(
|
| 86 |
+
{"open": open_, "high": high, "low": low, "close": close, "volume": vol}, index=index
|
| 87 |
+
)
|
| 88 |
+
|
| 89 |
+
|
| 90 |
+
def permute_bars(
|
| 91 |
+
df: pd.DataFrame,
|
| 92 |
+
rng: np.random.Generator,
|
| 93 |
+
method: str = "permute",
|
| 94 |
+
block: int = 20,
|
| 95 |
+
) -> pd.DataFrame:
|
| 96 |
+
"""Return a shuffled market with the same index and bar anatomy.
|
| 97 |
+
|
| 98 |
+
``permute`` reshuffles individual bars, destroying all serial structure.
|
| 99 |
+
``block`` resamples contiguous blocks with replacement, which preserves
|
| 100 |
+
short-horizon autocorrelation and volatility clustering — a harder null
|
| 101 |
+
that trend strategies deserve to be tested against.
|
| 102 |
+
"""
|
| 103 |
+
parts, anchor = _decompose(df)
|
| 104 |
+
m = parts.shape[0]
|
| 105 |
+
if m < 2:
|
| 106 |
+
return df.copy()
|
| 107 |
+
|
| 108 |
+
if method == "block":
|
| 109 |
+
size = max(2, min(int(block), m))
|
| 110 |
+
starts = rng.integers(0, m, size=int(np.ceil(m / size)))
|
| 111 |
+
order = np.concatenate([(np.arange(s, s + size) % m) for s in starts])[:m]
|
| 112 |
+
else:
|
| 113 |
+
order = rng.permutation(m)
|
| 114 |
+
|
| 115 |
+
return _rebuild(parts[order], anchor, df.index, df.iloc[0])
|
| 116 |
+
|
| 117 |
+
|
| 118 |
+
def permutation_test(
|
| 119 |
+
df: pd.DataFrame,
|
| 120 |
+
signal_fn: Callable[[pd.DataFrame], pd.Series],
|
| 121 |
+
n_permutations: int = 300,
|
| 122 |
+
method: str = "permute",
|
| 123 |
+
block: int = 20,
|
| 124 |
+
costs: Optional[CostModel] = None,
|
| 125 |
+
lag: int = 1,
|
| 126 |
+
max_leverage: float = 1.0,
|
| 127 |
+
allow_short: bool = True,
|
| 128 |
+
seed: int = 0,
|
| 129 |
+
observed: Optional[float] = None,
|
| 130 |
+
progress: Optional[Callable[[float, str], None]] = None,
|
| 131 |
+
) -> PermutationResult:
|
| 132 |
+
"""Run ``signal_fn`` against ``n_permutations`` shuffled markets.
|
| 133 |
+
|
| 134 |
+
``signal_fn`` must be the strategy's target-exposure generator; it is
|
| 135 |
+
re-evaluated on every synthetic market, which is the whole point — a rule
|
| 136 |
+
that only works because of the specific path it was tuned on will fall
|
| 137 |
+
apart here.
|
| 138 |
+
"""
|
| 139 |
+
costs = costs or CostModel()
|
| 140 |
+
rng = np.random.default_rng(seed)
|
| 141 |
+
|
| 142 |
+
def sharpe_on(frame: pd.DataFrame) -> float:
|
| 143 |
+
target = signal_fn(frame)
|
| 144 |
+
result = run_backtest(
|
| 145 |
+
frame,
|
| 146 |
+
target,
|
| 147 |
+
costs=costs,
|
| 148 |
+
lag=lag,
|
| 149 |
+
max_leverage=max_leverage,
|
| 150 |
+
allow_short=allow_short,
|
| 151 |
+
)
|
| 152 |
+
return result.sharpe
|
| 153 |
+
|
| 154 |
+
if observed is None:
|
| 155 |
+
observed = sharpe_on(df)
|
| 156 |
+
|
| 157 |
+
null = np.empty(n_permutations, dtype=float)
|
| 158 |
+
for i in range(n_permutations):
|
| 159 |
+
null[i] = sharpe_on(permute_bars(df, rng, method, block))
|
| 160 |
+
if progress is not None and (i % 25 == 0 or i == n_permutations - 1):
|
| 161 |
+
progress((i + 1) / n_permutations, f"Permutation {i + 1}/{n_permutations}")
|
| 162 |
+
|
| 163 |
+
# +1 in both places: the observed result is itself one draw from the null
|
| 164 |
+
# under H0, which keeps the test from ever reporting an impossible p = 0.
|
| 165 |
+
p_value = float((1 + np.sum(null >= observed)) / (n_permutations + 1))
|
| 166 |
+
|
| 167 |
+
return PermutationResult(
|
| 168 |
+
observed=float(observed),
|
| 169 |
+
null=null,
|
| 170 |
+
p_value=p_value,
|
| 171 |
+
method=method,
|
| 172 |
+
n_permutations=n_permutations,
|
| 173 |
+
)
|
| 174 |
+
|
| 175 |
+
|
| 176 |
+
def bootstrap_return_paths(returns: pd.Series, n: int = 500, seed: int = 0) -> np.ndarray:
|
| 177 |
+
"""Bootstrap terminal-wealth outcomes from a realised return stream.
|
| 178 |
+
|
| 179 |
+
Useful for the "how wide is the cone of outcomes?" chart — the same edge
|
| 180 |
+
can produce wildly different equity curves.
|
| 181 |
+
"""
|
| 182 |
+
arr = np.asarray(returns.dropna(), dtype=float)
|
| 183 |
+
if arr.size == 0:
|
| 184 |
+
return np.zeros((n, 1))
|
| 185 |
+
rng = np.random.default_rng(seed)
|
| 186 |
+
draws = rng.choice(arr, size=(n, arr.size), replace=True)
|
| 187 |
+
return np.cumprod(1.0 + draws, axis=1)
|
algotrader/validation/walkforward.py
ADDED
|
@@ -0,0 +1,233 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Walk-forward analysis.
|
| 2 |
+
|
| 3 |
+
Re-tune on a training window, trade the next window blind, roll forward. The
|
| 4 |
+
gap between in-sample and out-of-sample Sharpe is the honest estimate of how
|
| 5 |
+
much of the backtest was curve-fitting.
|
| 6 |
+
"""
|
| 7 |
+
|
| 8 |
+
from __future__ import annotations
|
| 9 |
+
|
| 10 |
+
from typing import Callable, Dict, List, Optional
|
| 11 |
+
|
| 12 |
+
import numpy as np
|
| 13 |
+
import pandas as pd
|
| 14 |
+
|
| 15 |
+
from ..engine import run_backtest
|
| 16 |
+
from ..strategies import Strategy
|
| 17 |
+
from ..types import CostModel
|
| 18 |
+
|
| 19 |
+
__all__ = ["walk_forward", "walk_forward_panel"]
|
| 20 |
+
|
| 21 |
+
|
| 22 |
+
def walk_forward(
|
| 23 |
+
df: pd.DataFrame,
|
| 24 |
+
strategy: Strategy,
|
| 25 |
+
n_folds: int = 5,
|
| 26 |
+
train_ratio: float = 0.7,
|
| 27 |
+
costs: Optional[CostModel] = None,
|
| 28 |
+
lag: int = 1,
|
| 29 |
+
max_leverage: float = 1.0,
|
| 30 |
+
allow_short: bool = True,
|
| 31 |
+
grid_limit: int = 40,
|
| 32 |
+
progress: Optional[Callable[[float, str], None]] = None,
|
| 33 |
+
) -> Dict[str, object]:
|
| 34 |
+
"""Roll a train/test split forward ``n_folds`` times.
|
| 35 |
+
|
| 36 |
+
Each fold picks the best parameters by in-sample Sharpe and reports what
|
| 37 |
+
those parameters then did out of sample. The stitched OOS returns are the
|
| 38 |
+
closest thing to a paper-trading record this repo can produce offline.
|
| 39 |
+
"""
|
| 40 |
+
costs = costs or CostModel()
|
| 41 |
+
grid = strategy.grid(limit=grid_limit)
|
| 42 |
+
n = len(df)
|
| 43 |
+
|
| 44 |
+
if n < 250 or n_folds < 2:
|
| 45 |
+
return {"folds": [], "note": "Not enough history for walk-forward analysis."}
|
| 46 |
+
|
| 47 |
+
# Each fold is a contiguous train+test block; blocks advance by test length.
|
| 48 |
+
block = int(n / (1 + (n_folds - 1) * (1 - train_ratio)))
|
| 49 |
+
block = min(block, n)
|
| 50 |
+
train_len = int(block * train_ratio)
|
| 51 |
+
test_len = block - train_len
|
| 52 |
+
if train_len < 100 or test_len < 20:
|
| 53 |
+
return {"folds": [], "note": "Not enough history for walk-forward analysis."}
|
| 54 |
+
|
| 55 |
+
folds: List[Dict[str, object]] = []
|
| 56 |
+
oos_returns: List[pd.Series] = []
|
| 57 |
+
|
| 58 |
+
for k in range(n_folds):
|
| 59 |
+
start = k * test_len
|
| 60 |
+
train = df.iloc[start : start + train_len]
|
| 61 |
+
test = df.iloc[start + train_len : start + train_len + test_len]
|
| 62 |
+
if len(test) < 20:
|
| 63 |
+
break
|
| 64 |
+
|
| 65 |
+
best_params, best_sharpe = None, -np.inf
|
| 66 |
+
for params in grid:
|
| 67 |
+
target = strategy.generate(train, params)
|
| 68 |
+
sr = run_backtest(
|
| 69 |
+
train, target, costs=costs, lag=lag,
|
| 70 |
+
max_leverage=max_leverage, allow_short=allow_short,
|
| 71 |
+
).sharpe
|
| 72 |
+
if sr > best_sharpe:
|
| 73 |
+
best_params, best_sharpe = params, sr
|
| 74 |
+
|
| 75 |
+
# Generate signals on train+test so indicators are warm at the fold
|
| 76 |
+
# boundary, then evaluate only the test slice.
|
| 77 |
+
combined = df.iloc[start : start + train_len + len(test)]
|
| 78 |
+
target = strategy.generate(combined, best_params).loc[test.index]
|
| 79 |
+
oos = run_backtest(
|
| 80 |
+
test, target, costs=costs, lag=lag,
|
| 81 |
+
max_leverage=max_leverage, allow_short=allow_short,
|
| 82 |
+
)
|
| 83 |
+
|
| 84 |
+
folds.append(
|
| 85 |
+
{
|
| 86 |
+
"fold": k + 1,
|
| 87 |
+
"train_start": str(train.index[0].date()),
|
| 88 |
+
"train_end": str(train.index[-1].date()),
|
| 89 |
+
"test_start": str(test.index[0].date()),
|
| 90 |
+
"test_end": str(test.index[-1].date()),
|
| 91 |
+
"params": best_params,
|
| 92 |
+
"is_sharpe": float(best_sharpe),
|
| 93 |
+
"oos_sharpe": float(oos.sharpe),
|
| 94 |
+
"oos_return": float(oos.metrics.get("total_return", 0.0)),
|
| 95 |
+
"oos_max_dd": float(oos.metrics.get("max_drawdown", 0.0)),
|
| 96 |
+
}
|
| 97 |
+
)
|
| 98 |
+
oos_returns.append(oos.returns)
|
| 99 |
+
|
| 100 |
+
if progress is not None:
|
| 101 |
+
progress((k + 1) / n_folds, f"Walk-forward fold {k + 1}/{n_folds}")
|
| 102 |
+
|
| 103 |
+
if not folds:
|
| 104 |
+
return {"folds": [], "note": "Not enough history for walk-forward analysis."}
|
| 105 |
+
|
| 106 |
+
is_sharpes = np.array([f["is_sharpe"] for f in folds], dtype=float)
|
| 107 |
+
oos_sharpes = np.array([f["oos_sharpe"] for f in folds], dtype=float)
|
| 108 |
+
stitched = pd.concat(oos_returns) if oos_returns else pd.Series(dtype=float)
|
| 109 |
+
stitched = stitched[~stitched.index.duplicated(keep="first")].sort_index()
|
| 110 |
+
|
| 111 |
+
mean_is = float(np.mean(is_sharpes))
|
| 112 |
+
mean_oos = float(np.mean(oos_sharpes))
|
| 113 |
+
|
| 114 |
+
return {
|
| 115 |
+
"folds": folds,
|
| 116 |
+
"mean_is_sharpe": mean_is,
|
| 117 |
+
"mean_oos_sharpe": mean_oos,
|
| 118 |
+
# 1.0 = the edge fully survived; 0.0 = it evaporated out of sample.
|
| 119 |
+
"efficiency": float(mean_oos / mean_is) if mean_is > 1e-9 else 0.0,
|
| 120 |
+
"oos_win_rate": float(np.mean(oos_sharpes > 0)),
|
| 121 |
+
# 1.0 means every fold chose different parameters -- a tuning process
|
| 122 |
+
# that cannot make up its mind is fitting noise.
|
| 123 |
+
"param_instability": float(
|
| 124 |
+
len({str(f["params"]) for f in folds}) / max(len(folds), 1)
|
| 125 |
+
),
|
| 126 |
+
"oos_returns": stitched,
|
| 127 |
+
"oos_equity": (1.0 + stitched).cumprod() if len(stitched) else stitched,
|
| 128 |
+
"note": "",
|
| 129 |
+
}
|
| 130 |
+
|
| 131 |
+
|
| 132 |
+
def walk_forward_panel(
|
| 133 |
+
panel,
|
| 134 |
+
strategy,
|
| 135 |
+
n_folds: int = 4,
|
| 136 |
+
train_ratio: float = 0.7,
|
| 137 |
+
costs: Optional[CostModel] = None,
|
| 138 |
+
lag: int = 1,
|
| 139 |
+
gross_leverage: float = 1.0,
|
| 140 |
+
allow_short: bool = True,
|
| 141 |
+
rebalance: str = "M",
|
| 142 |
+
grid_limit: int = 16,
|
| 143 |
+
progress: Optional[Callable[[float, str], None]] = None,
|
| 144 |
+
) -> Dict[str, object]:
|
| 145 |
+
"""Walk-forward for cross-sectional strategies over a :class:`Panel`.
|
| 146 |
+
|
| 147 |
+
Same contract as :func:`walk_forward`: tune on the training window, trade
|
| 148 |
+
the next window blind, roll on. Signals are generated over train+test
|
| 149 |
+
together so the indicators are warm at the fold boundary, then evaluated
|
| 150 |
+
only on the test slice -- the panel equivalent of the single-asset path.
|
| 151 |
+
"""
|
| 152 |
+
from ..portfolio import rebalance_schedule, run_portfolio_backtest
|
| 153 |
+
|
| 154 |
+
costs = costs or CostModel()
|
| 155 |
+
grid = strategy.grid(limit=grid_limit)
|
| 156 |
+
n = len(panel)
|
| 157 |
+
|
| 158 |
+
if n < 250 or n_folds < 2:
|
| 159 |
+
return {"folds": [], "note": "Not enough history for walk-forward analysis."}
|
| 160 |
+
|
| 161 |
+
block = min(int(n / (1 + (n_folds - 1) * (1 - train_ratio))), n)
|
| 162 |
+
train_len = int(block * train_ratio)
|
| 163 |
+
test_len = block - train_len
|
| 164 |
+
if train_len < 100 or test_len < 20:
|
| 165 |
+
return {"folds": [], "note": "Not enough history for walk-forward analysis."}
|
| 166 |
+
|
| 167 |
+
folds: List[Dict[str, object]] = []
|
| 168 |
+
oos_returns: List[pd.Series] = []
|
| 169 |
+
|
| 170 |
+
for k in range(n_folds):
|
| 171 |
+
start = k * test_len
|
| 172 |
+
train = panel.slice(panel.index[start], panel.index[min(start + train_len - 1, n - 1)])
|
| 173 |
+
test_start = start + train_len
|
| 174 |
+
if test_start + 20 > n:
|
| 175 |
+
break
|
| 176 |
+
test_end = min(test_start + test_len, n) - 1
|
| 177 |
+
test = panel.slice(panel.index[test_start], panel.index[test_end])
|
| 178 |
+
if len(test) < 20:
|
| 179 |
+
break
|
| 180 |
+
|
| 181 |
+
best_params, best_sharpe = None, -np.inf
|
| 182 |
+
for params in grid:
|
| 183 |
+
weights = strategy.generate(train, params)
|
| 184 |
+
sharpe = run_portfolio_backtest(
|
| 185 |
+
train, weights, costs=costs, lag=lag, gross_leverage=gross_leverage,
|
| 186 |
+
allow_short=allow_short, rebalance_on=rebalance_schedule(train.index, rebalance),
|
| 187 |
+
).sharpe
|
| 188 |
+
if sharpe > best_sharpe:
|
| 189 |
+
best_params, best_sharpe = params, sharpe
|
| 190 |
+
|
| 191 |
+
combined = panel.slice(panel.index[start], panel.index[test_end])
|
| 192 |
+
weights = strategy.generate(combined, best_params).loc[test.index]
|
| 193 |
+
oos = run_portfolio_backtest(
|
| 194 |
+
test, weights, costs=costs, lag=lag, gross_leverage=gross_leverage,
|
| 195 |
+
allow_short=allow_short, rebalance_on=rebalance_schedule(test.index, rebalance),
|
| 196 |
+
)
|
| 197 |
+
|
| 198 |
+
folds.append({
|
| 199 |
+
"fold": k + 1,
|
| 200 |
+
"train_start": str(train.index[0].date()),
|
| 201 |
+
"train_end": str(train.index[-1].date()),
|
| 202 |
+
"test_start": str(test.index[0].date()),
|
| 203 |
+
"test_end": str(test.index[-1].date()),
|
| 204 |
+
"params": best_params,
|
| 205 |
+
"is_sharpe": float(best_sharpe),
|
| 206 |
+
"oos_sharpe": float(oos.sharpe),
|
| 207 |
+
"oos_return": float(oos.metrics.get("total_return", 0.0)),
|
| 208 |
+
"oos_max_dd": float(oos.metrics.get("max_drawdown", 0.0)),
|
| 209 |
+
})
|
| 210 |
+
oos_returns.append(oos.returns)
|
| 211 |
+
if progress is not None:
|
| 212 |
+
progress((k + 1) / n_folds, f"Walk-forward fold {k + 1}/{n_folds}")
|
| 213 |
+
|
| 214 |
+
if not folds:
|
| 215 |
+
return {"folds": [], "note": "Not enough history for walk-forward analysis."}
|
| 216 |
+
|
| 217 |
+
is_sharpes = np.array([f["is_sharpe"] for f in folds], dtype=float)
|
| 218 |
+
oos_sharpes = np.array([f["oos_sharpe"] for f in folds], dtype=float)
|
| 219 |
+
stitched = pd.concat(oos_returns) if oos_returns else pd.Series(dtype=float)
|
| 220 |
+
stitched = stitched[~stitched.index.duplicated(keep="first")].sort_index()
|
| 221 |
+
mean_is, mean_oos = float(np.mean(is_sharpes)), float(np.mean(oos_sharpes))
|
| 222 |
+
|
| 223 |
+
return {
|
| 224 |
+
"folds": folds,
|
| 225 |
+
"mean_is_sharpe": mean_is,
|
| 226 |
+
"mean_oos_sharpe": mean_oos,
|
| 227 |
+
"efficiency": float(mean_oos / mean_is) if mean_is > 1e-9 else 0.0,
|
| 228 |
+
"oos_win_rate": float(np.mean(oos_sharpes > 0)),
|
| 229 |
+
"param_instability": float(len({str(f["params"]) for f in folds}) / max(len(folds), 1)),
|
| 230 |
+
"oos_returns": stitched,
|
| 231 |
+
"oos_equity": (1.0 + stitched).cumprod() if len(stitched) else stitched,
|
| 232 |
+
"note": "",
|
| 233 |
+
}
|
algotrader/verdict.py
ADDED
|
@@ -0,0 +1,211 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Turn a pile of statistics into one number and one sentence.
|
| 2 |
+
|
| 3 |
+
The Reality Score is deliberately harsh. Most published backtests would score
|
| 4 |
+
below 40, and that is the point: the score exists to be screenshotted.
|
| 5 |
+
"""
|
| 6 |
+
|
| 7 |
+
from __future__ import annotations
|
| 8 |
+
|
| 9 |
+
from typing import Dict, List, Optional
|
| 10 |
+
|
| 11 |
+
import numpy as np
|
| 12 |
+
|
| 13 |
+
__all__ = ["reality_score", "GRADES"]
|
| 14 |
+
|
| 15 |
+
GRADES = [
|
| 16 |
+
(85, "A", "Survives everything we threw at it"),
|
| 17 |
+
(70, "B", "Probably a real edge, with caveats"),
|
| 18 |
+
(55, "C", "Ambiguous — could go either way"),
|
| 19 |
+
(40, "D", "Mostly luck"),
|
| 20 |
+
(0, "F", "Indistinguishable from randomness"),
|
| 21 |
+
]
|
| 22 |
+
|
| 23 |
+
WEIGHTS = {
|
| 24 |
+
"significance": 0.30,
|
| 25 |
+
"selection": 0.25,
|
| 26 |
+
"walk_forward": 0.20,
|
| 27 |
+
"overfitting": 0.15,
|
| 28 |
+
"robustness": 0.10,
|
| 29 |
+
}
|
| 30 |
+
|
| 31 |
+
|
| 32 |
+
def _ramp(value: float, good: float, bad: float) -> float:
|
| 33 |
+
"""Linear 0-100 score where ``good`` maps to 100 and ``bad`` maps to 0."""
|
| 34 |
+
if not np.isfinite(value):
|
| 35 |
+
return 50.0
|
| 36 |
+
if good == bad:
|
| 37 |
+
return 50.0
|
| 38 |
+
scaled = (value - bad) / (good - bad)
|
| 39 |
+
return float(np.clip(scaled, 0.0, 1.0) * 100.0)
|
| 40 |
+
|
| 41 |
+
|
| 42 |
+
def reality_score(
|
| 43 |
+
metrics: Dict[str, float],
|
| 44 |
+
benchmark_metrics: Dict[str, float],
|
| 45 |
+
p_value: Optional[float] = None,
|
| 46 |
+
dsr: Optional[float] = None,
|
| 47 |
+
pbo: Optional[float] = None,
|
| 48 |
+
wf_efficiency: Optional[float] = None,
|
| 49 |
+
wf_win_rate: Optional[float] = None,
|
| 50 |
+
cost_stress_ratio: Optional[float] = None,
|
| 51 |
+
benchmark_correlation: Optional[float] = None,
|
| 52 |
+
attribution: Optional[Dict[str, object]] = None,
|
| 53 |
+
survivorship: Optional[object] = None,
|
| 54 |
+
benchmark_name: str = "Buy & hold",
|
| 55 |
+
permutation_label: str = "shuffled, structure-free markets",
|
| 56 |
+
) -> Dict[str, object]:
|
| 57 |
+
"""Combine the validation panel into a 0-100 score, a grade and warnings.
|
| 58 |
+
|
| 59 |
+
``attribution`` and ``survivorship`` are the portfolio-only inputs; both are
|
| 60 |
+
optional so the single-asset path is unaffected.
|
| 61 |
+
"""
|
| 62 |
+
components: Dict[str, float] = {}
|
| 63 |
+
|
| 64 |
+
components["significance"] = _ramp(p_value, good=0.01, bad=0.50) if p_value is not None else 50.0
|
| 65 |
+
components["selection"] = float(np.clip(dsr, 0.0, 1.0) * 100.0) if dsr is not None else 50.0
|
| 66 |
+
|
| 67 |
+
if wf_efficiency is not None:
|
| 68 |
+
wf = _ramp(wf_efficiency, good=0.8, bad=-0.2)
|
| 69 |
+
if wf_win_rate is not None:
|
| 70 |
+
wf = 0.7 * wf + 0.3 * float(np.clip(wf_win_rate, 0.0, 1.0) * 100.0)
|
| 71 |
+
components["walk_forward"] = wf
|
| 72 |
+
else:
|
| 73 |
+
components["walk_forward"] = 50.0
|
| 74 |
+
|
| 75 |
+
components["overfitting"] = _ramp(pbo, good=0.05, bad=0.50) if pbo is not None and np.isfinite(pbo) else 50.0
|
| 76 |
+
components["robustness"] = (
|
| 77 |
+
_ramp(cost_stress_ratio, good=0.8, bad=0.0) if cost_stress_ratio is not None else 50.0
|
| 78 |
+
)
|
| 79 |
+
|
| 80 |
+
score = float(sum(components[k] * w for k, w in WEIGHTS.items()))
|
| 81 |
+
|
| 82 |
+
flags: List[str] = []
|
| 83 |
+
n_trades = int(metrics.get("n_trades", 0))
|
| 84 |
+
sharpe = float(metrics.get("sharpe", 0.0))
|
| 85 |
+
total_return = float(metrics.get("total_return", 0.0))
|
| 86 |
+
max_dd = float(metrics.get("max_drawdown", 0.0))
|
| 87 |
+
bench_sharpe = float(benchmark_metrics.get("sharpe", 0.0))
|
| 88 |
+
|
| 89 |
+
# The score asks "is the measured edge real?" — if the strategy lost money,
|
| 90 |
+
# there is no edge to validate, whatever the statistics say about it.
|
| 91 |
+
if total_return <= 0:
|
| 92 |
+
flags.append(
|
| 93 |
+
f"The strategy lost money over the test period ({total_return:.1%}). "
|
| 94 |
+
"There is no edge here to validate."
|
| 95 |
+
)
|
| 96 |
+
score = min(score, 50.0)
|
| 97 |
+
if total_return <= 0 < sharpe:
|
| 98 |
+
flags.append(
|
| 99 |
+
"Positive Sharpe with a negative total return: the average bar was profitable but "
|
| 100 |
+
"the compounding was not. Volatility drag ate the arithmetic edge."
|
| 101 |
+
)
|
| 102 |
+
if max_dd < -0.5:
|
| 103 |
+
flags.append(
|
| 104 |
+
f"Peak-to-trough drawdown of {max_dd:.0%} — an account running this would have been "
|
| 105 |
+
"closed long before the recovery arrived."
|
| 106 |
+
)
|
| 107 |
+
score = min(score, 60.0)
|
| 108 |
+
|
| 109 |
+
if n_trades < 20:
|
| 110 |
+
flags.append(
|
| 111 |
+
f"Only {n_trades} position changes — with this few decisions, "
|
| 112 |
+
"the result is a handful of coin flips, not a track record."
|
| 113 |
+
)
|
| 114 |
+
score = min(score, 55.0)
|
| 115 |
+
if p_value is not None and p_value > 0.10:
|
| 116 |
+
flags.append(
|
| 117 |
+
f"Permutation p-value is {p_value:.2f}: roughly {p_value * 100:.0f}% of "
|
| 118 |
+
f"{permutation_label} did this well or better."
|
| 119 |
+
)
|
| 120 |
+
if dsr is not None and dsr < 0.5:
|
| 121 |
+
flags.append(
|
| 122 |
+
f"Deflated Sharpe is {dsr:.2f} — once you account for how many variants were tried, "
|
| 123 |
+
"the edge does not clear the selection-bias bar."
|
| 124 |
+
)
|
| 125 |
+
if pbo is not None and np.isfinite(pbo) and pbo > 0.3:
|
| 126 |
+
flags.append(
|
| 127 |
+
f"Probability of backtest overfitting is {pbo:.0%}: the in-sample winner "
|
| 128 |
+
"usually lands in the bottom half out of sample."
|
| 129 |
+
)
|
| 130 |
+
if wf_efficiency is not None and wf_efficiency < 0.3:
|
| 131 |
+
flags.append(
|
| 132 |
+
f"Walk-forward efficiency is {wf_efficiency:.0%} — most of the in-sample Sharpe "
|
| 133 |
+
"does not survive re-tuning and trading forward."
|
| 134 |
+
)
|
| 135 |
+
if cost_stress_ratio is not None and cost_stress_ratio < 0.5:
|
| 136 |
+
flags.append(
|
| 137 |
+
"Tripling trading costs removes more than half the Sharpe. The edge is "
|
| 138 |
+
"smaller than the friction it has to pay."
|
| 139 |
+
)
|
| 140 |
+
if benchmark_correlation is not None and benchmark_correlation > 0.95:
|
| 141 |
+
flags.append(
|
| 142 |
+
f"Returns are {benchmark_correlation:.0%} correlated with {benchmark_name.lower()} — "
|
| 143 |
+
"this is mostly a repackaged long position."
|
| 144 |
+
)
|
| 145 |
+
if sharpe < bench_sharpe:
|
| 146 |
+
flags.append(
|
| 147 |
+
f"{benchmark_name} beat it on risk-adjusted return "
|
| 148 |
+
f"({bench_sharpe:.2f} vs {sharpe:.2f} Sharpe)."
|
| 149 |
+
)
|
| 150 |
+
if float(metrics.get("turnover_ann", 0.0)) > 100:
|
| 151 |
+
flags.append(
|
| 152 |
+
f"Annual turnover of {metrics.get('turnover_ann', 0):.0f}x is far beyond what "
|
| 153 |
+
"retail execution can absorb without moving the modelled fills."
|
| 154 |
+
)
|
| 155 |
+
|
| 156 |
+
# Portfolio-only checks. Style exposure you could buy in an ETF is not
|
| 157 |
+
# alpha, and a universe with no failures in it is not a universe.
|
| 158 |
+
if attribution and attribution.get("available"):
|
| 159 |
+
if not attribution.get("alpha_significant"):
|
| 160 |
+
flags.append(
|
| 161 |
+
f"Style regression leaves no significant alpha (t = "
|
| 162 |
+
f"{attribution.get('alpha_t_stat', 0):.1f}); the factors explain "
|
| 163 |
+
f"{attribution.get('r_squared', 0):.0%} of returns"
|
| 164 |
+
+ (
|
| 165 |
+
f", mostly {attribution['dominant_factor']} exposure."
|
| 166 |
+
if attribution.get("dominant_factor")
|
| 167 |
+
else "."
|
| 168 |
+
)
|
| 169 |
+
)
|
| 170 |
+
score = min(score, 65.0)
|
| 171 |
+
if float(attribution.get("r_squared", 0.0)) > 0.9:
|
| 172 |
+
flags.append(
|
| 173 |
+
"Over 90% of the return variation is explained by simple style factors — "
|
| 174 |
+
"this book is a repackaged index."
|
| 175 |
+
)
|
| 176 |
+
if survivorship is not None and getattr(survivorship, "biased", False):
|
| 177 |
+
flags.append(getattr(survivorship, "note", "Universe appears survivorship-biased."))
|
| 178 |
+
score = min(score, 60.0)
|
| 179 |
+
|
| 180 |
+
score = float(np.clip(score, 0.0, 100.0))
|
| 181 |
+
grade, headline = next((g, h) for threshold, g, h in GRADES if score >= threshold)
|
| 182 |
+
|
| 183 |
+
if score >= 70:
|
| 184 |
+
verdict = (
|
| 185 |
+
f"Grade {grade}. {headline}. The edge is still there after the permutation test, "
|
| 186 |
+
"after charging for every variant tried, and after walking it forward."
|
| 187 |
+
)
|
| 188 |
+
elif score >= 55:
|
| 189 |
+
verdict = (
|
| 190 |
+
f"Grade {grade}. {headline}. Parts of the panel hold up and parts do not — "
|
| 191 |
+
"this is the zone where more data, not more tuning, is what settles it."
|
| 192 |
+
)
|
| 193 |
+
elif score >= 40:
|
| 194 |
+
verdict = (
|
| 195 |
+
f"Grade {grade}. {headline}. The backtest looks better than the evidence supports; "
|
| 196 |
+
"the gap between the two is selection bias."
|
| 197 |
+
)
|
| 198 |
+
else:
|
| 199 |
+
verdict = (
|
| 200 |
+
f"Grade {grade}. {headline}. The null — {permutation_label} — produces results "
|
| 201 |
+
"like this often enough that there is nothing here to trade."
|
| 202 |
+
)
|
| 203 |
+
|
| 204 |
+
return {
|
| 205 |
+
"score": round(score, 1),
|
| 206 |
+
"grade": grade,
|
| 207 |
+
"headline": headline,
|
| 208 |
+
"verdict": verdict,
|
| 209 |
+
"components": {k: round(v, 1) for k, v in components.items()},
|
| 210 |
+
"flags": flags,
|
| 211 |
+
}
|
app.py
ADDED
|
@@ -0,0 +1,849 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Backtest Reality Check — the Hugging Face Space entrypoint.
|
| 2 |
+
|
| 3 |
+
Point it at a ticker and a trading rule. It runs the backtest, then spends the
|
| 4 |
+
rest of its time trying to prove the result was luck: shuffled markets,
|
| 5 |
+
selection-bias deflation, walk-forward, and a cost stress test.
|
| 6 |
+
|
| 7 |
+
Run locally with ``python app.py``.
|
| 8 |
+
"""
|
| 9 |
+
|
| 10 |
+
from __future__ import annotations
|
| 11 |
+
|
| 12 |
+
import logging
|
| 13 |
+
import os
|
| 14 |
+
from typing import Dict, List
|
| 15 |
+
|
| 16 |
+
import gradio as gr
|
| 17 |
+
import pandas as pd
|
| 18 |
+
|
| 19 |
+
from algotrader import __version__
|
| 20 |
+
from algotrader.charts import (
|
| 21 |
+
arena_chart,
|
| 22 |
+
attribution_chart,
|
| 23 |
+
cross_permutation_chart,
|
| 24 |
+
drawdown_chart,
|
| 25 |
+
empty_figure,
|
| 26 |
+
equity_chart,
|
| 27 |
+
exposure_chart,
|
| 28 |
+
permutation_chart,
|
| 29 |
+
score_chart,
|
| 30 |
+
walkforward_chart,
|
| 31 |
+
weights_chart,
|
| 32 |
+
)
|
| 33 |
+
from algotrader.cross_sectional import XS_REGISTRY, get_xs_strategy
|
| 34 |
+
from algotrader.data import DEFAULT_UNIVERSE
|
| 35 |
+
from algotrader.lab import LabConfig, run_arena, run_lab
|
| 36 |
+
from algotrader.portfolio_lab import DEFAULT_UNIVERSE as PORTFOLIO_UNIVERSE
|
| 37 |
+
from algotrader.portfolio_lab import PortfolioLabConfig, run_portfolio_lab
|
| 38 |
+
from algotrader.strategies import REGISTRY, get_strategy
|
| 39 |
+
|
| 40 |
+
logging.basicConfig(level=logging.INFO, format="%(levelname)s %(name)s: %(message)s")
|
| 41 |
+
logger = logging.getLogger("app")
|
| 42 |
+
|
| 43 |
+
MAX_PARAMS = 3
|
| 44 |
+
MAX_XS_PARAMS = 4
|
| 45 |
+
STRATEGY_CHOICES = [(s.name, key) for key, s in REGISTRY.items()]
|
| 46 |
+
XS_STRATEGY_CHOICES = [(s.name, key) for key, s in XS_REGISTRY.items()]
|
| 47 |
+
|
| 48 |
+
GRADE_COLORS = {
|
| 49 |
+
"A": "#0ca30c",
|
| 50 |
+
"B": "#3987e5",
|
| 51 |
+
"C": "#fab219",
|
| 52 |
+
"D": "#ec835a",
|
| 53 |
+
"F": "#d03b3b",
|
| 54 |
+
}
|
| 55 |
+
|
| 56 |
+
CSS = """
|
| 57 |
+
.gradio-container { max-width: 1280px !important; }
|
| 58 |
+
#hero h1 { font-size: 2.4rem; line-height: 1.1; margin: 0 0 .4rem 0; letter-spacing: -.02em; }
|
| 59 |
+
#hero p { color: #c3c2b7; margin: 0; font-size: 1.05rem; max-width: 60ch; }
|
| 60 |
+
.score-card {
|
| 61 |
+
display: flex; gap: 24px; align-items: center; padding: 22px 24px;
|
| 62 |
+
border: 1px solid rgba(255,255,255,.10); border-radius: 14px; background: #1a1a19;
|
| 63 |
+
}
|
| 64 |
+
.score-badge {
|
| 65 |
+
min-width: 132px; text-align: center; padding: 14px 10px; border-radius: 12px;
|
| 66 |
+
background: #0d0d0d; border: 1px solid rgba(255,255,255,.10);
|
| 67 |
+
}
|
| 68 |
+
.score-badge .grade { font-size: 3.2rem; font-weight: 700; line-height: 1; }
|
| 69 |
+
.score-badge .num { font-size: .95rem; color: #898781; margin-top: 6px; }
|
| 70 |
+
.score-body h3 { margin: 0 0 6px 0; font-size: 1.15rem; color: #fff; }
|
| 71 |
+
.score-body p { margin: 0; color: #c3c2b7; line-height: 1.55; }
|
| 72 |
+
.tiles { display: grid; grid-template-columns: repeat(auto-fit, minmax(132px,1fr)); gap: 10px; margin-top: 14px; }
|
| 73 |
+
.tile { padding: 12px 14px; border: 1px solid rgba(255,255,255,.10); border-radius: 10px; background: #1a1a19; }
|
| 74 |
+
.tile .label { font-size: .72rem; text-transform: uppercase; letter-spacing: .06em; color: #898781; }
|
| 75 |
+
.tile .value { font-size: 1.5rem; font-weight: 600; color: #fff; margin-top: 4px; }
|
| 76 |
+
.tile .sub { font-size: .75rem; color: #898781; margin-top: 2px; }
|
| 77 |
+
.flags { margin-top: 14px; padding: 0; list-style: none; }
|
| 78 |
+
.flags li {
|
| 79 |
+
padding: 9px 12px; margin-bottom: 7px; border-radius: 8px; background: #1a1a19;
|
| 80 |
+
border-left: 3px solid #ec835a; color: #c3c2b7; font-size: .9rem; line-height: 1.5;
|
| 81 |
+
}
|
| 82 |
+
.provenance { font-size: .82rem; color: #898781; margin-top: 10px; }
|
| 83 |
+
.provenance.sim { color: #fab219; }
|
| 84 |
+
"""
|
| 85 |
+
|
| 86 |
+
|
| 87 |
+
def _fmt_pct(value: float) -> str:
|
| 88 |
+
return f"{value * 100:,.1f}%"
|
| 89 |
+
|
| 90 |
+
|
| 91 |
+
def param_controls(strategy_key: str):
|
| 92 |
+
"""Re-label the shared sliders to match the selected strategy."""
|
| 93 |
+
strategy = get_strategy(strategy_key)
|
| 94 |
+
updates = []
|
| 95 |
+
for i in range(MAX_PARAMS):
|
| 96 |
+
if i < len(strategy.params):
|
| 97 |
+
spec = strategy.params[i]
|
| 98 |
+
lo = spec.minimum if spec.minimum is not None else min(spec.grid)
|
| 99 |
+
hi = spec.maximum if spec.maximum is not None else max(spec.grid)
|
| 100 |
+
updates.append(
|
| 101 |
+
gr.update(
|
| 102 |
+
visible=True,
|
| 103 |
+
label=spec.label,
|
| 104 |
+
value=spec.cast(spec.default),
|
| 105 |
+
minimum=lo,
|
| 106 |
+
maximum=hi,
|
| 107 |
+
step=spec.step or (1 if spec.kind == "int" else 0.1),
|
| 108 |
+
)
|
| 109 |
+
)
|
| 110 |
+
else:
|
| 111 |
+
updates.append(gr.update(visible=False))
|
| 112 |
+
return tuple(updates)
|
| 113 |
+
|
| 114 |
+
|
| 115 |
+
def collect_params(strategy_key: str, *values) -> Dict[str, float]:
|
| 116 |
+
strategy = get_strategy(strategy_key)
|
| 117 |
+
return {spec.name: spec.cast(values[i]) for i, spec in enumerate(strategy.params[:MAX_PARAMS])}
|
| 118 |
+
|
| 119 |
+
|
| 120 |
+
def _score_card(report) -> str:
|
| 121 |
+
v = report.verdict
|
| 122 |
+
color = GRADE_COLORS.get(v["grade"], "#898781")
|
| 123 |
+
market = report.market
|
| 124 |
+
provenance = (
|
| 125 |
+
f'<div class="provenance">Data: {market.source} · {market.symbol} · '
|
| 126 |
+
f"{market.start.date()} to {market.end.date()} · {len(market.df):,} bars</div>"
|
| 127 |
+
if market.is_real
|
| 128 |
+
else f'<div class="provenance sim">⚠ {market.note}</div>'
|
| 129 |
+
)
|
| 130 |
+
flags = "".join(f"<li>{f}</li>" for f in v["flags"])
|
| 131 |
+
flags_html = f'<ul class="flags">{flags}</ul>' if flags else ""
|
| 132 |
+
|
| 133 |
+
return f"""
|
| 134 |
+
<div class="score-card">
|
| 135 |
+
<div class="score-badge">
|
| 136 |
+
<div class="grade" style="color:{color}">{v['grade']}</div>
|
| 137 |
+
<div class="num">{v['score']} / 100</div>
|
| 138 |
+
</div>
|
| 139 |
+
<div class="score-body">
|
| 140 |
+
<h3>{report.strategy.name} on {market.symbol}</h3>
|
| 141 |
+
<p>{v['verdict']}</p>
|
| 142 |
+
</div>
|
| 143 |
+
</div>
|
| 144 |
+
{flags_html}
|
| 145 |
+
{provenance}
|
| 146 |
+
"""
|
| 147 |
+
|
| 148 |
+
|
| 149 |
+
def _tiles(report) -> str:
|
| 150 |
+
m = report.backtest.metrics
|
| 151 |
+
b = report.backtest.benchmark_metrics
|
| 152 |
+
perm = report.permutation
|
| 153 |
+
p_text = f"{perm.p_value:.3f}" if perm else "—"
|
| 154 |
+
p_sub = "vs shuffled markets" if perm else "test skipped"
|
| 155 |
+
pbo = report.pbo.get("pbo")
|
| 156 |
+
pbo_text = f"{pbo:.0%}" if pbo is not None and pbo == pbo else "n/a"
|
| 157 |
+
|
| 158 |
+
cells = [
|
| 159 |
+
("Total return", _fmt_pct(m.get("total_return", 0)), f"buy & hold {_fmt_pct(b.get('total_return', 0))}"),
|
| 160 |
+
("CAGR", _fmt_pct(m.get("cagr", 0)), f"over {m.get('years', 0):.1f} years"),
|
| 161 |
+
("Sharpe", f"{m.get('sharpe', 0):.2f}", f"buy & hold {b.get('sharpe', 0):.2f}"),
|
| 162 |
+
("Max drawdown", _fmt_pct(m.get("max_drawdown", 0)), f"{m.get('time_under_water_yrs', 0):.1f}y under water"),
|
| 163 |
+
("p-value", p_text, p_sub),
|
| 164 |
+
("Deflated Sharpe", f"{report.dsr.get('dsr', 0):.2f}", f"after {report.trials.get('n', 1)} variants"),
|
| 165 |
+
("Overfit prob.", pbo_text, "in-sample winner fails OOS"),
|
| 166 |
+
("Trades", f"{int(m.get('n_trades', 0)):,}", f"{m.get('turnover_ann', 0):.0f}x turnover/yr"),
|
| 167 |
+
]
|
| 168 |
+
tiles = "".join(
|
| 169 |
+
f'<div class="tile"><div class="label">{label}</div>'
|
| 170 |
+
f'<div class="value">{value}</div><div class="sub">{sub}</div></div>'
|
| 171 |
+
for label, value, sub in cells
|
| 172 |
+
)
|
| 173 |
+
return f'<div class="tiles">{tiles}</div>'
|
| 174 |
+
|
| 175 |
+
|
| 176 |
+
def xs_param_controls(strategy_key: str):
|
| 177 |
+
"""Re-label the shared portfolio sliders for the selected strategy."""
|
| 178 |
+
strategy = get_xs_strategy(strategy_key)
|
| 179 |
+
updates = []
|
| 180 |
+
for i in range(MAX_XS_PARAMS):
|
| 181 |
+
if i < len(strategy.params):
|
| 182 |
+
spec = strategy.params[i]
|
| 183 |
+
updates.append(
|
| 184 |
+
gr.update(
|
| 185 |
+
visible=True,
|
| 186 |
+
label=spec.label,
|
| 187 |
+
value=spec.cast(spec.default),
|
| 188 |
+
minimum=spec.minimum if spec.minimum is not None else min(spec.grid),
|
| 189 |
+
maximum=spec.maximum if spec.maximum is not None else max(spec.grid),
|
| 190 |
+
step=spec.step or (1 if spec.kind == "int" else 0.05),
|
| 191 |
+
)
|
| 192 |
+
)
|
| 193 |
+
else:
|
| 194 |
+
updates.append(gr.update(visible=False))
|
| 195 |
+
return tuple(updates)
|
| 196 |
+
|
| 197 |
+
|
| 198 |
+
def _portfolio_card(report) -> str:
|
| 199 |
+
v = report.verdict
|
| 200 |
+
color = GRADE_COLORS.get(v["grade"], "#898781")
|
| 201 |
+
panel = report.panel
|
| 202 |
+
survivorship = report.survivorship
|
| 203 |
+
|
| 204 |
+
provenance = (
|
| 205 |
+
f'<div class="provenance">{len(panel.symbols)} symbols · '
|
| 206 |
+
f"{panel.index[0].date()} to {panel.index[-1].date()} · {len(panel):,} bars · "
|
| 207 |
+
f"rebalance {report.config.rebalance}</div>"
|
| 208 |
+
)
|
| 209 |
+
if panel.note:
|
| 210 |
+
provenance += f'<div class="provenance sim">⚠ {panel.note}</div>'
|
| 211 |
+
# The survivorship note is already in the flag list when it is a problem;
|
| 212 |
+
# repeating it under the tiles just makes the card noisy.
|
| 213 |
+
|
| 214 |
+
flags = "".join(f"<li>{f}</li>" for f in v["flags"])
|
| 215 |
+
flags_html = f'<ul class="flags">{flags}</ul>' if flags else ""
|
| 216 |
+
|
| 217 |
+
return f"""
|
| 218 |
+
<div class="score-card">
|
| 219 |
+
<div class="score-badge">
|
| 220 |
+
<div class="grade" style="color:{color}">{v['grade']}</div>
|
| 221 |
+
<div class="num">{v['score']} / 100</div>
|
| 222 |
+
</div>
|
| 223 |
+
<div class="score-body">
|
| 224 |
+
<h3>{report.strategy.name} across {len(panel.symbols)} names</h3>
|
| 225 |
+
<p>{v['verdict']}</p>
|
| 226 |
+
</div>
|
| 227 |
+
</div>
|
| 228 |
+
{flags_html}
|
| 229 |
+
{provenance}
|
| 230 |
+
"""
|
| 231 |
+
|
| 232 |
+
|
| 233 |
+
def _portfolio_tiles(report) -> str:
|
| 234 |
+
m = report.backtest.metrics
|
| 235 |
+
b = report.backtest.benchmark_metrics
|
| 236 |
+
perm = report.permutation
|
| 237 |
+
attribution = report.attribution
|
| 238 |
+
pbo = report.pbo.get("pbo")
|
| 239 |
+
|
| 240 |
+
cells = [
|
| 241 |
+
("Total return", _fmt_pct(m.get("total_return", 0)), f"equal weight {_fmt_pct(b.get('total_return', 0))}"),
|
| 242 |
+
("Sharpe", f"{m.get('sharpe', 0):.2f}", f"equal weight {b.get('sharpe', 0):.2f}"),
|
| 243 |
+
("Max drawdown", _fmt_pct(m.get("max_drawdown", 0)), f"{m.get('time_under_water_yrs', 0):.1f}y under water"),
|
| 244 |
+
("Name-shuffle p", f"{perm.p_value:.3f}" if perm else "—", "vs same book, random names"),
|
| 245 |
+
("Deflated Sharpe", f"{report.dsr.get('dsr', 0):.2f}", f"after {report.trials.get('n', 1)} variants"),
|
| 246 |
+
(
|
| 247 |
+
"Style alpha",
|
| 248 |
+
f"{attribution.get('alpha_annual', 0) * 100:,.1f}%" if attribution.get("available") else "n/a",
|
| 249 |
+
f"t = {attribution.get('alpha_t_stat', 0):.1f}" if attribution.get("available") else "not available",
|
| 250 |
+
),
|
| 251 |
+
("Overfit prob.", f"{pbo:.0%}" if pbo is not None and pbo == pbo else "n/a", "in-sample winner fails OOS"),
|
| 252 |
+
("Gross / net", f"{m.get('gross_exposure', 0):.2f}", f"net {m.get('net_exposure', 0):+.2f}"),
|
| 253 |
+
("Avg positions", f"{m.get('avg_positions', 0):.1f}", f"{m.get('turnover_ann', 0):.1f}x turnover/yr"),
|
| 254 |
+
(
|
| 255 |
+
"Survivorship",
|
| 256 |
+
f"{report.survivorship.survival_rate:.0%}",
|
| 257 |
+
f"{report.survivorship.n_delisted} of {report.survivorship.n_symbols} delisted",
|
| 258 |
+
),
|
| 259 |
+
]
|
| 260 |
+
tiles = "".join(
|
| 261 |
+
f'<div class="tile"><div class="label">{label}</div>'
|
| 262 |
+
f'<div class="value">{value}</div><div class="sub">{sub}</div></div>'
|
| 263 |
+
for label, value, sub in cells
|
| 264 |
+
)
|
| 265 |
+
return f'<div class="tiles">{tiles}</div>'
|
| 266 |
+
|
| 267 |
+
|
| 268 |
+
def _portfolio_detail(report) -> str:
|
| 269 |
+
attribution, survivorship, wf = report.attribution, report.survivorship, report.walkforward
|
| 270 |
+
lines = ["### Reading the evidence", ""]
|
| 271 |
+
|
| 272 |
+
if report.permutation:
|
| 273 |
+
lines += [
|
| 274 |
+
f"**Did it pick the right names?** We rebuilt the book "
|
| 275 |
+
f"{report.permutation.n_permutations} times, keeping every date's gross exposure, net "
|
| 276 |
+
f"exposure and position count exactly as they were, and only randomising *which* asset "
|
| 277 |
+
f"got which weight. The real book's Sharpe of {report.permutation.observed:.2f} sits at "
|
| 278 |
+
f"the {report.permutation.percentile:.0f}th percentile of that null "
|
| 279 |
+
f"(p = {report.permutation.p_value:.3f}). This test deliberately ignores trading costs: "
|
| 280 |
+
"a shuffled book churns far more than a real one, and charging it for that would "
|
| 281 |
+
"flatter the strategy for reasons unrelated to skill.",
|
| 282 |
+
"",
|
| 283 |
+
]
|
| 284 |
+
if attribution.get("available"):
|
| 285 |
+
lines += [f"**Is it alpha or is it beta?** {attribution['note']}", ""]
|
| 286 |
+
lines += [f"**Survivorship.** {survivorship.note}", ""]
|
| 287 |
+
if survivorship.delisted_symbols:
|
| 288 |
+
lines += [f"Stopped trading during the sample: `{'`, `'.join(survivorship.delisted_symbols)}`.", ""]
|
| 289 |
+
if wf.get("folds"):
|
| 290 |
+
lines += [
|
| 291 |
+
f"**Walk-forward.** Across {len(wf['folds'])} folds the tuned in-sample Sharpe averaged "
|
| 292 |
+
f"{wf.get('mean_is_sharpe', 0):.2f} against {wf.get('mean_oos_sharpe', 0):.2f} blind out "
|
| 293 |
+
f"of sample — {wf.get('efficiency', 0):.0%} efficiency, with "
|
| 294 |
+
f"{wf.get('oos_win_rate', 0):.0%} of folds profitable.",
|
| 295 |
+
"",
|
| 296 |
+
]
|
| 297 |
+
lines += [
|
| 298 |
+
f"**Costs.** Sharpe is {report.cost_stress.get('sharpe_1x', 0):.2f} at the modelled friction "
|
| 299 |
+
f"and {report.cost_stress.get('sharpe_3x', 0):.2f} at triple it "
|
| 300 |
+
f"({report.cost_stress.get('ratio', 0):.0%} retained), on turnover of "
|
| 301 |
+
f"{report.backtest.metrics.get('turnover_ann', 0):.1f}x a year.",
|
| 302 |
+
"",
|
| 303 |
+
"> Past performance, simulated or otherwise, does not predict future returns. "
|
| 304 |
+
"This is research tooling, not investment advice.",
|
| 305 |
+
]
|
| 306 |
+
return "\n".join(lines)
|
| 307 |
+
|
| 308 |
+
|
| 309 |
+
def analyse_portfolio(
|
| 310 |
+
symbols: str,
|
| 311 |
+
start: str,
|
| 312 |
+
end: str,
|
| 313 |
+
strategy_key: str,
|
| 314 |
+
p1: float,
|
| 315 |
+
p2: float,
|
| 316 |
+
p3: float,
|
| 317 |
+
p4: float,
|
| 318 |
+
rebalance: str,
|
| 319 |
+
commission: float,
|
| 320 |
+
slippage: float,
|
| 321 |
+
allow_short: bool,
|
| 322 |
+
gross_leverage: float,
|
| 323 |
+
max_weight: float,
|
| 324 |
+
n_permutations: int,
|
| 325 |
+
progress=gr.Progress(),
|
| 326 |
+
):
|
| 327 |
+
"""Portfolio Lab handler. Like the single-asset one, it never raises into the UI."""
|
| 328 |
+
try:
|
| 329 |
+
universe = [s.strip().upper() for s in (symbols or "").replace("\n", ",").split(",") if s.strip()]
|
| 330 |
+
if len(universe) < 4:
|
| 331 |
+
raise ValueError(
|
| 332 |
+
"A cross-sectional strategy needs at least 4 symbols — it ranks names against "
|
| 333 |
+
"each other, and there is nothing to rank in a list this short."
|
| 334 |
+
)
|
| 335 |
+
strategy = get_xs_strategy(strategy_key)
|
| 336 |
+
values = (p1, p2, p3, p4)
|
| 337 |
+
params = {
|
| 338 |
+
spec.name: spec.cast(values[i])
|
| 339 |
+
for i, spec in enumerate(strategy.params[:MAX_XS_PARAMS])
|
| 340 |
+
}
|
| 341 |
+
cfg = PortfolioLabConfig(
|
| 342 |
+
symbols=universe,
|
| 343 |
+
start=start or "2015-01-01",
|
| 344 |
+
end=end or None,
|
| 345 |
+
source="yahoo",
|
| 346 |
+
strategy=strategy_key,
|
| 347 |
+
params=params,
|
| 348 |
+
commission_bps=float(commission),
|
| 349 |
+
slippage_bps=float(slippage),
|
| 350 |
+
allow_short=bool(allow_short),
|
| 351 |
+
gross_leverage=float(gross_leverage),
|
| 352 |
+
max_weight=float(max_weight) if max_weight else None,
|
| 353 |
+
rebalance=rebalance,
|
| 354 |
+
n_permutations=int(n_permutations),
|
| 355 |
+
)
|
| 356 |
+
report = run_portfolio_lab(cfg, progress=lambda f, m: progress(f, desc=m))
|
| 357 |
+
except Exception as exc: # noqa: BLE001
|
| 358 |
+
logger.exception("Portfolio lab run failed")
|
| 359 |
+
message = (
|
| 360 |
+
f'<div class="score-card"><div class="score-body"><h3>Could not run that</h3>'
|
| 361 |
+
f"<p>{exc}</p></div></div>"
|
| 362 |
+
)
|
| 363 |
+
blank = empty_figure("No results.")
|
| 364 |
+
return message, "", blank, blank, blank, blank, blank, ""
|
| 365 |
+
|
| 366 |
+
return (
|
| 367 |
+
_portfolio_card(report),
|
| 368 |
+
_portfolio_tiles(report),
|
| 369 |
+
cross_permutation_chart(report),
|
| 370 |
+
equity_chart(report, benchmark_label="Equal weight"),
|
| 371 |
+
attribution_chart(report.attribution),
|
| 372 |
+
weights_chart(report),
|
| 373 |
+
score_chart(report.verdict, significance_label="Picks the right names"),
|
| 374 |
+
_portfolio_detail(report),
|
| 375 |
+
)
|
| 376 |
+
|
| 377 |
+
|
| 378 |
+
def _detail_markdown(report) -> str:
|
| 379 |
+
dsr, wf, pbo, stress = report.dsr, report.walkforward, report.pbo, report.cost_stress
|
| 380 |
+
mtr = dsr.get("min_track_record_years", float("inf"))
|
| 381 |
+
mtr_text = f"{mtr:.1f} years" if mtr == mtr and mtr != float("inf") else "never, at this effect size"
|
| 382 |
+
|
| 383 |
+
lines = [
|
| 384 |
+
"### Reading the evidence",
|
| 385 |
+
"",
|
| 386 |
+
f"**Selection bias.** {report.trials.get('n', 1)} parameter variants of "
|
| 387 |
+
f"*{report.strategy.name}* were backtested. The luckiest skill-free variant of that many "
|
| 388 |
+
f"would be expected to show an annualised Sharpe of about "
|
| 389 |
+
f"**{dsr.get('threshold_sr_annual', 0):.2f}** on its own. Yours was "
|
| 390 |
+
f"**{report.backtest.sharpe:.2f}**, which puts the deflated probability of a real edge at "
|
| 391 |
+
f"**{dsr.get('dsr', 0):.0%}**.",
|
| 392 |
+
"",
|
| 393 |
+
f"**Track record needed.** To call this Sharpe significant at 95% confidence given its "
|
| 394 |
+
f"skew ({dsr.get('skew', 0):.2f}) and kurtosis ({dsr.get('kurtosis', 0):.1f}), you would need "
|
| 395 |
+
f"about **{mtr_text}** of live returns.",
|
| 396 |
+
"",
|
| 397 |
+
f"**Costs.** At the modelled friction the Sharpe is {stress.get('sharpe_1x', 0):.2f}. "
|
| 398 |
+
f"Triple the costs and it becomes {stress.get('sharpe_3x', 0):.2f} "
|
| 399 |
+
f"({stress.get('ratio', 0):.0%} retained).",
|
| 400 |
+
"",
|
| 401 |
+
]
|
| 402 |
+
|
| 403 |
+
if wf.get("folds"):
|
| 404 |
+
lines += [
|
| 405 |
+
f"**Walk-forward.** Across {len(wf['folds'])} folds the tuned in-sample Sharpe averaged "
|
| 406 |
+
f"{wf.get('mean_is_sharpe', 0):.2f} and the blind out-of-sample Sharpe averaged "
|
| 407 |
+
f"{wf.get('mean_oos_sharpe', 0):.2f} — an efficiency of {wf.get('efficiency', 0):.0%}. "
|
| 408 |
+
f"{wf.get('oos_win_rate', 0):.0%} of folds were profitable out of sample, and the tuner "
|
| 409 |
+
f"picked a different parameter set in {wf.get('param_instability', 0):.0%} of them.",
|
| 410 |
+
"",
|
| 411 |
+
]
|
| 412 |
+
if pbo.get("note"):
|
| 413 |
+
lines += [f"**Overfitting.** {pbo['note']}", ""]
|
| 414 |
+
elif pbo.get("pbo") == pbo.get("pbo"):
|
| 415 |
+
lines += [
|
| 416 |
+
f"**Overfitting (CSCV).** Over {pbo.get('n_combinations', 0)} in/out splits of "
|
| 417 |
+
f"{pbo.get('n_strategies', 0)} variants, the in-sample winner landed in the bottom half "
|
| 418 |
+
f"out of sample **{pbo.get('pbo', 0):.0%}** of the time, and lost money outright "
|
| 419 |
+
f"{pbo.get('prob_oos_loss', 0):.0%} of the time. The most frequently selected variant was "
|
| 420 |
+
f"`{pbo.get('most_selected_label', 'n/a')}` "
|
| 421 |
+
f"({pbo.get('selection_stability', 0):.0%} of splits).",
|
| 422 |
+
"",
|
| 423 |
+
]
|
| 424 |
+
|
| 425 |
+
lines += [
|
| 426 |
+
"> Past performance, simulated or otherwise, does not predict future returns. "
|
| 427 |
+
"This is research tooling, not investment advice.",
|
| 428 |
+
]
|
| 429 |
+
return "\n".join(lines)
|
| 430 |
+
|
| 431 |
+
|
| 432 |
+
def analyse(
|
| 433 |
+
symbol: str,
|
| 434 |
+
start: str,
|
| 435 |
+
end: str,
|
| 436 |
+
strategy_key: str,
|
| 437 |
+
p1: float,
|
| 438 |
+
p2: float,
|
| 439 |
+
p3: float,
|
| 440 |
+
commission: float,
|
| 441 |
+
slippage: float,
|
| 442 |
+
allow_short: bool,
|
| 443 |
+
n_permutations: int,
|
| 444 |
+
perm_method: str,
|
| 445 |
+
wf_folds: int,
|
| 446 |
+
progress=gr.Progress(),
|
| 447 |
+
):
|
| 448 |
+
"""Main Lab handler. Never raises into the UI — it returns a readable message."""
|
| 449 |
+
try:
|
| 450 |
+
cfg = LabConfig(
|
| 451 |
+
symbol=symbol or "SPY",
|
| 452 |
+
start=start or "2015-01-01",
|
| 453 |
+
end=end or None,
|
| 454 |
+
source="yahoo",
|
| 455 |
+
strategy=strategy_key,
|
| 456 |
+
params=collect_params(strategy_key, p1, p2, p3),
|
| 457 |
+
commission_bps=float(commission),
|
| 458 |
+
slippage_bps=float(slippage),
|
| 459 |
+
allow_short=bool(allow_short),
|
| 460 |
+
n_permutations=int(n_permutations),
|
| 461 |
+
permutation_method="block" if perm_method.startswith("Block") else "permute",
|
| 462 |
+
wf_folds=int(wf_folds),
|
| 463 |
+
)
|
| 464 |
+
report = run_lab(cfg, progress=lambda f, m: progress(f, desc=m))
|
| 465 |
+
except Exception as exc: # noqa: BLE001 - the UI must always say something useful
|
| 466 |
+
logger.exception("Lab run failed")
|
| 467 |
+
message = f'<div class="score-card"><div class="score-body"><h3>Could not run that</h3><p>{exc}</p></div></div>'
|
| 468 |
+
blank = empty_figure("No results.")
|
| 469 |
+
return message, "", blank, blank, blank, blank, blank, blank, ""
|
| 470 |
+
|
| 471 |
+
return (
|
| 472 |
+
_score_card(report),
|
| 473 |
+
_tiles(report),
|
| 474 |
+
permutation_chart(report),
|
| 475 |
+
equity_chart(report),
|
| 476 |
+
drawdown_chart(report),
|
| 477 |
+
exposure_chart(report),
|
| 478 |
+
walkforward_chart(report),
|
| 479 |
+
score_chart(report.verdict),
|
| 480 |
+
_detail_markdown(report),
|
| 481 |
+
)
|
| 482 |
+
|
| 483 |
+
|
| 484 |
+
def race(symbol: str, start: str, allow_short: bool, n_permutations: int, progress=gr.Progress()):
|
| 485 |
+
try:
|
| 486 |
+
cfg = LabConfig(
|
| 487 |
+
symbol=symbol or "SPY",
|
| 488 |
+
start=start or "2015-01-01",
|
| 489 |
+
source="yahoo",
|
| 490 |
+
allow_short=bool(allow_short),
|
| 491 |
+
)
|
| 492 |
+
table, market, _ = run_arena(
|
| 493 |
+
cfg,
|
| 494 |
+
n_permutations=int(n_permutations),
|
| 495 |
+
progress=lambda f, m: progress(f, desc=m),
|
| 496 |
+
)
|
| 497 |
+
except Exception as exc: # noqa: BLE001
|
| 498 |
+
logger.exception("Arena run failed")
|
| 499 |
+
return pd.DataFrame({"Error": [str(exc)]}), empty_figure("No results."), ""
|
| 500 |
+
|
| 501 |
+
display = table.copy()
|
| 502 |
+
for col in ("Return", "CAGR", "MaxDD"):
|
| 503 |
+
display[col] = display[col].map(lambda v: f"{v * 100:,.1f}%")
|
| 504 |
+
for col in ("Sharpe", "DSR", "Evidence"):
|
| 505 |
+
display[col] = display[col].map(lambda v: f"{v:.2f}")
|
| 506 |
+
display["p-value"] = display["p-value"].map(lambda v: "—" if v != v else f"{v:.3f}")
|
| 507 |
+
display = display.drop(columns=["key"])
|
| 508 |
+
|
| 509 |
+
provenance = (
|
| 510 |
+
f"Data: {market.source} · {market.symbol} · {market.start.date()} to {market.end.date()}"
|
| 511 |
+
if market.is_real
|
| 512 |
+
else f"⚠ {market.note}"
|
| 513 |
+
)
|
| 514 |
+
note = (
|
| 515 |
+
f"{provenance}\n\nRanked by **evidence** — `(1 − p) × deflated Sharpe` — not by return. "
|
| 516 |
+
"*Buy & Hold* and *Coin Flip* are in the field on purpose: a leaderboard without a "
|
| 517 |
+
"control group is marketing, not measurement."
|
| 518 |
+
)
|
| 519 |
+
return display, arena_chart(table), note
|
| 520 |
+
|
| 521 |
+
|
| 522 |
+
HOW_IT_WORKS = """
|
| 523 |
+
## Why most backtests are wrong
|
| 524 |
+
|
| 525 |
+
A backtest is a measurement taken with a ruler you built after seeing the thing you
|
| 526 |
+
are measuring. Four failure modes do almost all the damage, and this Space tests for
|
| 527 |
+
each one.
|
| 528 |
+
|
| 529 |
+
### 1. The market had no structure to find — permutation test
|
| 530 |
+
|
| 531 |
+
We take the real price series and shuffle it: each bar's gap, high, low, body and
|
| 532 |
+
volume are kept intact, but their **order** is destroyed. The result is a market with
|
| 533 |
+
the same volatility and the same fat tails, and no exploitable structure whatsoever.
|
| 534 |
+
Then we re-run *your exact rule* on hundreds of these shuffled markets.
|
| 535 |
+
|
| 536 |
+
If your Sharpe sits inside that cloud of results, your rule found nothing that a
|
| 537 |
+
coin-flip market would not also have handed it. The **p-value** is the share of
|
| 538 |
+
shuffled markets that did as well or better.
|
| 539 |
+
|
| 540 |
+
*Block mode* resamples contiguous chunks instead of single bars, preserving
|
| 541 |
+
short-horizon momentum and volatility clustering. It is a harder null, and trend
|
| 542 |
+
strategies should be held to it.
|
| 543 |
+
|
| 544 |
+
### 2. You tried 200 things and reported the best — Deflated Sharpe Ratio
|
| 545 |
+
|
| 546 |
+
If you test 200 worthless strategies, the best of them will show a Sharpe near 1.0
|
| 547 |
+
purely by chance. The **Deflated Sharpe Ratio** (Bailey & López de Prado, 2014) works
|
| 548 |
+
out what the luckiest of *N* skill-free variants would have scored, and asks whether
|
| 549 |
+
yours beats that bar — with an extra penalty for negative skew and fat tails, the
|
| 550 |
+
return shapes that flatter naive Sharpe ratios.
|
| 551 |
+
|
| 552 |
+
This Space counts the whole parameter grid as trials, because that is what a
|
| 553 |
+
researcher would really have run.
|
| 554 |
+
|
| 555 |
+
### 3. The parameters were fitted to the past — PBO and walk-forward
|
| 556 |
+
|
| 557 |
+
**Probability of Backtest Overfitting** (CSCV) cuts the timeline into chunks, and for
|
| 558 |
+
every way of splitting them half in-sample and half out-of-sample, checks whether the
|
| 559 |
+
in-sample winner stayed a winner. If the winner lands in the bottom half about half
|
| 560 |
+
the time, PBO ≈ 50% and your selection process has no skill at all.
|
| 561 |
+
|
| 562 |
+
**Walk-forward** re-tunes on a training window and trades the next window blind,
|
| 563 |
+
rolling forward. Efficiency is out-of-sample Sharpe over in-sample Sharpe: 100% means
|
| 564 |
+
the edge survived intact, 0% means it was entirely curve-fit.
|
| 565 |
+
|
| 566 |
+
### 4. The edge is smaller than the costs — stress test
|
| 567 |
+
|
| 568 |
+
Every result here is net of commission and slippage charged on exposure changes, plus
|
| 569 |
+
a borrow fee on short positions. We then re-run at **triple** the friction. A real edge
|
| 570 |
+
degrades; a fake one disappears.
|
| 571 |
+
|
| 572 |
+
---
|
| 573 |
+
|
| 574 |
+
## The Reality Score
|
| 575 |
+
|
| 576 |
+
| Weight | Component | What it measures |
|
| 577 |
+
|---:|---|---|
|
| 578 |
+
| 30% | Significance | How far outside the shuffled-market null the result sits |
|
| 579 |
+
| 25% | Selection | Deflated Sharpe — does it clear the best-of-N bar |
|
| 580 |
+
| 20% | Walk-forward | How much of the tuned Sharpe survived trading forward |
|
| 581 |
+
| 15% | Overfitting | 1 − PBO, from combinatorially symmetric cross-validation |
|
| 582 |
+
| 10% | Robustness | Sharpe retained when costs triple |
|
| 583 |
+
|
| 584 |
+
Grades: **A** ≥ 85 · **B** ≥ 70 · **C** ≥ 55 · **D** ≥ 40 · **F** below 40.
|
| 585 |
+
|
| 586 |
+
The scale is deliberately harsh. Most strategies people post online score below 40,
|
| 587 |
+
and the honest response to that is not to soften the scale.
|
| 588 |
+
|
| 589 |
+
## No look-ahead, by construction
|
| 590 |
+
|
| 591 |
+
A strategy emits a target exposure at each bar's close using only data up to that
|
| 592 |
+
bar. The engine holds `position[t] = target[t - lag]` with `lag ≥ 1`, so a signal
|
| 593 |
+
computed on Tuesday's close cannot earn Tuesday's move. That is the single line where
|
| 594 |
+
look-ahead could enter, and the test suite asserts it directly.
|
| 595 |
+
|
| 596 |
+
## Use it from Python
|
| 597 |
+
|
| 598 |
+
```python
|
| 599 |
+
from algotrader import LabConfig, run_lab
|
| 600 |
+
|
| 601 |
+
report = run_lab(LabConfig(symbol="SPY", strategy="sma_cross", params={"fast": 20, "slow": 100}))
|
| 602 |
+
print(report.verdict["grade"], report.verdict["score"])
|
| 603 |
+
print(report.permutation.p_value, report.dsr["dsr"], report.pbo["pbo"])
|
| 604 |
+
```
|
| 605 |
+
|
| 606 |
+
Or from the command line:
|
| 607 |
+
|
| 608 |
+
```bash
|
| 609 |
+
python -m algotrader.cli lab --symbol SPY --strategy donchian_breakout --permutations 500
|
| 610 |
+
python -m algotrader.cli arena --symbol BTC-USD
|
| 611 |
+
```
|
| 612 |
+
|
| 613 |
+
---
|
| 614 |
+
|
| 615 |
+
*Research tooling, not investment advice. Nothing here is a recommendation to trade.*
|
| 616 |
+
"""
|
| 617 |
+
|
| 618 |
+
|
| 619 |
+
# Gradio 6 moved `css` and `theme` from the Blocks constructor to launch().
|
| 620 |
+
# Spaces pin their own version, so pass them wherever the installed one wants.
|
| 621 |
+
_GRADIO_MAJOR = int(gr.__version__.split(".")[0])
|
| 622 |
+
_STYLE_KWARGS = {"css": CSS, "theme": gr.themes.Base()}
|
| 623 |
+
_BLOCKS_KWARGS = {} if _GRADIO_MAJOR >= 6 else _STYLE_KWARGS
|
| 624 |
+
# Gradio 6 also dropped launch(show_api=...).
|
| 625 |
+
_LAUNCH_KWARGS = dict(_STYLE_KWARGS) if _GRADIO_MAJOR >= 6 else {"show_api": False}
|
| 626 |
+
|
| 627 |
+
|
| 628 |
+
def build_app() -> gr.Blocks:
|
| 629 |
+
with gr.Blocks(title="Backtest Reality Check", **_BLOCKS_KWARGS) as demo:
|
| 630 |
+
with gr.Column(elem_id="hero"):
|
| 631 |
+
gr.HTML(
|
| 632 |
+
"<h1>Backtest Reality Check</h1>"
|
| 633 |
+
"<p>Your backtest is probably lying to you. Pick a market and a trading rule — "
|
| 634 |
+
"this runs it, then spends the rest of its effort trying to prove the result "
|
| 635 |
+
"was luck.</p>"
|
| 636 |
+
)
|
| 637 |
+
|
| 638 |
+
with gr.Tabs():
|
| 639 |
+
with gr.Tab("The Lab"):
|
| 640 |
+
with gr.Row():
|
| 641 |
+
with gr.Column(scale=1):
|
| 642 |
+
symbol = gr.Dropdown(
|
| 643 |
+
choices=DEFAULT_UNIVERSE, value="SPY", label="Ticker",
|
| 644 |
+
allow_custom_value=True,
|
| 645 |
+
info="Any Yahoo Finance symbol. Falls back to a simulated market if offline.",
|
| 646 |
+
)
|
| 647 |
+
with gr.Row():
|
| 648 |
+
start = gr.Textbox(value="2015-01-01", label="Start", scale=1)
|
| 649 |
+
end = gr.Textbox(value="", label="End (blank = today)", scale=1)
|
| 650 |
+
|
| 651 |
+
strategy = gr.Dropdown(
|
| 652 |
+
choices=STRATEGY_CHOICES, value="sma_cross", label="Strategy"
|
| 653 |
+
)
|
| 654 |
+
strategy_note = gr.Markdown(get_strategy("sma_cross").description)
|
| 655 |
+
|
| 656 |
+
param_sliders = [
|
| 657 |
+
gr.Slider(label=f"Parameter {i + 1}", visible=False, minimum=0, maximum=100)
|
| 658 |
+
for i in range(MAX_PARAMS)
|
| 659 |
+
]
|
| 660 |
+
|
| 661 |
+
with gr.Accordion("Costs and testing", open=False):
|
| 662 |
+
commission = gr.Slider(0, 20, value=1, step=0.5, label="Commission (bps per trade)")
|
| 663 |
+
slippage = gr.Slider(0, 50, value=2, step=0.5, label="Slippage (bps per trade)")
|
| 664 |
+
allow_short = gr.Checkbox(value=True, label="Allow short positions")
|
| 665 |
+
n_perms = gr.Slider(
|
| 666 |
+
0, 1000, value=250, step=50, label="Shuffled markets to test against",
|
| 667 |
+
info="More is stricter and slower. 250 is plenty for a first look.",
|
| 668 |
+
)
|
| 669 |
+
perm_method = gr.Radio(
|
| 670 |
+
["Shuffle bars (standard)", "Block bootstrap (harder)"],
|
| 671 |
+
value="Shuffle bars (standard)", label="Null market",
|
| 672 |
+
)
|
| 673 |
+
wf_folds = gr.Slider(2, 8, value=5, step=1, label="Walk-forward folds")
|
| 674 |
+
|
| 675 |
+
run_button = gr.Button("Run reality check", variant="primary", size="lg")
|
| 676 |
+
|
| 677 |
+
gr.Examples(
|
| 678 |
+
label="Or try one of these",
|
| 679 |
+
examples=[
|
| 680 |
+
["SPY", "sma_cross"],
|
| 681 |
+
["BTC-USD", "donchian_breakout"],
|
| 682 |
+
["NVDA", "rsi_reversion"],
|
| 683 |
+
["QQQ", "momentum"],
|
| 684 |
+
["SPY", "coin_flip"],
|
| 685 |
+
],
|
| 686 |
+
inputs=[symbol, strategy],
|
| 687 |
+
)
|
| 688 |
+
|
| 689 |
+
with gr.Column(scale=2):
|
| 690 |
+
verdict_html = gr.HTML(
|
| 691 |
+
'<div class="score-card"><div class="score-body">'
|
| 692 |
+
"<h3>Nothing tested yet</h3><p>Pick a market and a rule, then hit "
|
| 693 |
+
"<b>Run reality check</b>. A full run is a few seconds.</p>"
|
| 694 |
+
"</div></div>"
|
| 695 |
+
)
|
| 696 |
+
tiles_html = gr.HTML("")
|
| 697 |
+
|
| 698 |
+
# The headline test gets the full width — it is the whole point.
|
| 699 |
+
perm_plot = gr.Plot(value=empty_figure("The headline test appears here.", height=320))
|
| 700 |
+
equity_plot = gr.Plot(value=empty_figure())
|
| 701 |
+
with gr.Row():
|
| 702 |
+
dd_plot = gr.Plot(value=empty_figure(height=240))
|
| 703 |
+
exposure_plot = gr.Plot(value=empty_figure(height=200))
|
| 704 |
+
with gr.Row():
|
| 705 |
+
wf_plot = gr.Plot(value=empty_figure(height=300))
|
| 706 |
+
components_plot = gr.Plot(value=empty_figure(height=260))
|
| 707 |
+
detail_md = gr.Markdown("")
|
| 708 |
+
|
| 709 |
+
strategy.change(
|
| 710 |
+
fn=param_controls, inputs=strategy, outputs=param_sliders
|
| 711 |
+
).then(
|
| 712 |
+
fn=lambda k: get_strategy(k).description, inputs=strategy, outputs=strategy_note
|
| 713 |
+
)
|
| 714 |
+
|
| 715 |
+
run_button.click(
|
| 716 |
+
fn=analyse,
|
| 717 |
+
inputs=[
|
| 718 |
+
symbol, start, end, strategy, *param_sliders,
|
| 719 |
+
commission, slippage, allow_short, n_perms, perm_method, wf_folds,
|
| 720 |
+
],
|
| 721 |
+
outputs=[
|
| 722 |
+
verdict_html, tiles_html, perm_plot, equity_plot,
|
| 723 |
+
dd_plot, exposure_plot, wf_plot, components_plot, detail_md,
|
| 724 |
+
],
|
| 725 |
+
)
|
| 726 |
+
|
| 727 |
+
with gr.Tab("Portfolio"):
|
| 728 |
+
gr.Markdown(
|
| 729 |
+
"Cross-sectional strategies rank names against each other, so they get a "
|
| 730 |
+
"harder null: we keep every date's gross exposure, net exposure and position "
|
| 731 |
+
"count exactly as they were and randomise only **which name got which "
|
| 732 |
+
"weight**. A book that beats that is picking names. One that doesn't was "
|
| 733 |
+
"being paid for style exposure you can buy in an ETF — which the factor "
|
| 734 |
+
"regression below measures directly."
|
| 735 |
+
)
|
| 736 |
+
with gr.Row():
|
| 737 |
+
with gr.Column(scale=1):
|
| 738 |
+
xs_symbols = gr.Textbox(
|
| 739 |
+
value=", ".join(PORTFOLIO_UNIVERSE),
|
| 740 |
+
label="Universe",
|
| 741 |
+
lines=3,
|
| 742 |
+
info="Comma-separated. At least 4 names — fewer cannot be ranked.",
|
| 743 |
+
)
|
| 744 |
+
with gr.Row():
|
| 745 |
+
xs_start = gr.Textbox(value="2015-01-01", label="Start", scale=1)
|
| 746 |
+
xs_end = gr.Textbox(value="", label="End", scale=1)
|
| 747 |
+
|
| 748 |
+
xs_strategy = gr.Dropdown(
|
| 749 |
+
choices=XS_STRATEGY_CHOICES, value="xs_momentum", label="Strategy"
|
| 750 |
+
)
|
| 751 |
+
xs_note = gr.Markdown(get_xs_strategy("xs_momentum").description)
|
| 752 |
+
xs_sliders = [
|
| 753 |
+
gr.Slider(label=f"Parameter {i + 1}", visible=False, minimum=0, maximum=100)
|
| 754 |
+
for i in range(MAX_XS_PARAMS)
|
| 755 |
+
]
|
| 756 |
+
xs_rebalance = gr.Radio(
|
| 757 |
+
["D", "W", "M", "Q"], value="M", label="Rebalance",
|
| 758 |
+
info="Daily rebalancing of a real book is rarely affordable.",
|
| 759 |
+
)
|
| 760 |
+
|
| 761 |
+
with gr.Accordion("Costs, limits and testing", open=False):
|
| 762 |
+
xs_commission = gr.Slider(0, 20, value=1, step=0.5, label="Commission (bps)")
|
| 763 |
+
xs_slippage = gr.Slider(0, 50, value=2, step=0.5, label="Slippage (bps)")
|
| 764 |
+
xs_short = gr.Checkbox(value=True, label="Allow short positions")
|
| 765 |
+
xs_leverage = gr.Slider(0.1, 3.0, value=1.0, step=0.1, label="Gross leverage cap")
|
| 766 |
+
xs_maxw = gr.Slider(0.0, 1.0, value=0.25, step=0.05, label="Max weight per name")
|
| 767 |
+
xs_perms = gr.Slider(
|
| 768 |
+
0, 500, value=150, step=25, label="Name shuffles",
|
| 769 |
+
)
|
| 770 |
+
|
| 771 |
+
xs_button = gr.Button("Run portfolio check", variant="primary", size="lg")
|
| 772 |
+
|
| 773 |
+
with gr.Column(scale=2):
|
| 774 |
+
xs_verdict = gr.HTML(
|
| 775 |
+
'<div class="score-card"><div class="score-body">'
|
| 776 |
+
"<h3>Nothing tested yet</h3><p>Pick a universe and a ranking rule, then "
|
| 777 |
+
"hit <b>Run portfolio check</b>.</p></div></div>"
|
| 778 |
+
)
|
| 779 |
+
xs_tiles = gr.HTML("")
|
| 780 |
+
|
| 781 |
+
xs_perm_plot = gr.Plot(value=empty_figure("The name-shuffle test appears here.", height=320))
|
| 782 |
+
xs_equity_plot = gr.Plot(value=empty_figure())
|
| 783 |
+
with gr.Row():
|
| 784 |
+
xs_attr_plot = gr.Plot(value=empty_figure(height=280))
|
| 785 |
+
xs_weights_plot = gr.Plot(value=empty_figure(height=240))
|
| 786 |
+
xs_components_plot = gr.Plot(value=empty_figure(height=260))
|
| 787 |
+
xs_detail = gr.Markdown("")
|
| 788 |
+
|
| 789 |
+
xs_strategy.change(
|
| 790 |
+
fn=xs_param_controls, inputs=xs_strategy, outputs=xs_sliders
|
| 791 |
+
).then(
|
| 792 |
+
fn=lambda k: get_xs_strategy(k).description, inputs=xs_strategy, outputs=xs_note
|
| 793 |
+
)
|
| 794 |
+
xs_button.click(
|
| 795 |
+
fn=analyse_portfolio,
|
| 796 |
+
inputs=[
|
| 797 |
+
xs_symbols, xs_start, xs_end, xs_strategy, *xs_sliders,
|
| 798 |
+
xs_rebalance, xs_commission, xs_slippage, xs_short,
|
| 799 |
+
xs_leverage, xs_maxw, xs_perms,
|
| 800 |
+
],
|
| 801 |
+
outputs=[
|
| 802 |
+
xs_verdict, xs_tiles, xs_perm_plot, xs_equity_plot,
|
| 803 |
+
xs_attr_plot, xs_weights_plot, xs_components_plot, xs_detail,
|
| 804 |
+
],
|
| 805 |
+
)
|
| 806 |
+
|
| 807 |
+
with gr.Tab("Arena"):
|
| 808 |
+
gr.Markdown(
|
| 809 |
+
"Race every strategy on the same market, ranked by **evidence** rather than "
|
| 810 |
+
"return. Buy & hold and a coin flip stay in the field as controls."
|
| 811 |
+
)
|
| 812 |
+
with gr.Row():
|
| 813 |
+
arena_symbol = gr.Dropdown(
|
| 814 |
+
choices=DEFAULT_UNIVERSE, value="SPY", label="Ticker", allow_custom_value=True
|
| 815 |
+
)
|
| 816 |
+
arena_start = gr.Textbox(value="2015-01-01", label="Start")
|
| 817 |
+
arena_short = gr.Checkbox(value=True, label="Allow shorts")
|
| 818 |
+
arena_perms = gr.Slider(0, 400, value=120, step=20, label="Shuffled markets per strategy")
|
| 819 |
+
arena_button = gr.Button("Run the arena", variant="primary")
|
| 820 |
+
arena_note = gr.Markdown("")
|
| 821 |
+
arena_table = gr.Dataframe(interactive=False, wrap=True)
|
| 822 |
+
arena_plot = gr.Plot(value=empty_figure(height=380))
|
| 823 |
+
|
| 824 |
+
arena_button.click(
|
| 825 |
+
fn=race,
|
| 826 |
+
inputs=[arena_symbol, arena_start, arena_short, arena_perms],
|
| 827 |
+
outputs=[arena_table, arena_plot, arena_note],
|
| 828 |
+
)
|
| 829 |
+
|
| 830 |
+
with gr.Tab("How it works"):
|
| 831 |
+
gr.Markdown(HOW_IT_WORKS)
|
| 832 |
+
|
| 833 |
+
gr.Markdown(
|
| 834 |
+
f"<sub>algotrader {__version__} · Apache-2.0 · "
|
| 835 |
+
"Research tooling, not investment advice.</sub>"
|
| 836 |
+
)
|
| 837 |
+
|
| 838 |
+
demo.load(fn=param_controls, inputs=strategy, outputs=param_sliders)
|
| 839 |
+
demo.load(fn=xs_param_controls, inputs=xs_strategy, outputs=xs_sliders)
|
| 840 |
+
|
| 841 |
+
return demo
|
| 842 |
+
|
| 843 |
+
|
| 844 |
+
if __name__ == "__main__":
|
| 845 |
+
build_app().queue(max_size=24).launch(
|
| 846 |
+
server_name="0.0.0.0",
|
| 847 |
+
server_port=int(os.environ.get("PORT", 7860)),
|
| 848 |
+
**_LAUNCH_KWARGS,
|
| 849 |
+
)
|
config.yaml
CHANGED
|
@@ -1,11 +1,11 @@
|
|
| 1 |
# Configuration file for the agentic AI trading system
|
| 2 |
data_source:
|
| 3 |
-
type: '
|
| 4 |
path: 'data/market_data.csv'
|
| 5 |
|
| 6 |
trading:
|
| 7 |
symbol: 'AAPL'
|
| 8 |
-
timeframe: '
|
| 9 |
capital: 100000
|
| 10 |
|
| 11 |
risk:
|
|
@@ -29,11 +29,17 @@ alpaca:
|
|
| 29 |
websocket_url: 'wss://stream.data.alpaca.markets/v2/iex' # WebSocket URL
|
| 30 |
account_type: 'paper' # 'paper' or 'live'
|
| 31 |
|
| 32 |
-
# Yahoo Finance (
|
| 33 |
# Unofficial API, typically delayed; 1m lookback is ~7 days.
|
| 34 |
yahoo:
|
| 35 |
poll_interval_seconds: 60
|
| 36 |
-
auto_adjust
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 37 |
start_date: '2024-01-01'
|
| 38 |
end_date: '2026-12-31'
|
| 39 |
|
|
|
|
| 1 |
# Configuration file for the agentic AI trading system
|
| 2 |
data_source:
|
| 3 |
+
type: 'yahoo'
|
| 4 |
path: 'data/market_data.csv'
|
| 5 |
|
| 6 |
trading:
|
| 7 |
symbol: 'AAPL'
|
| 8 |
+
timeframe: '1d'
|
| 9 |
capital: 100000
|
| 10 |
|
| 11 |
risk:
|
|
|
|
| 29 |
websocket_url: 'wss://stream.data.alpaca.markets/v2/iex' # WebSocket URL
|
| 30 |
account_type: 'paper' # 'paper' or 'live'
|
| 31 |
|
| 32 |
+
# Yahoo Finance (default ingest). Unofficial API, typically delayed; 1m lookback is ~7 days.
|
| 33 |
# Unofficial API, typically delayed; 1m lookback is ~7 days.
|
| 34 |
yahoo:
|
| 35 |
poll_interval_seconds: 60
|
| 36 |
+
# Keep true. With auto_adjust off, Yahoo returns raw Close and every stock
|
| 37 |
+
# split reads as a crash (NVDA's June 2024 10:1 becomes a -90% bar).
|
| 38 |
+
auto_adjust: true
|
| 39 |
+
# Yahoo returns the still-forming period as an ordinary row. Emitting it would
|
| 40 |
+
# trade on a close that has not happened yet.
|
| 41 |
+
emit_incomplete_bars: false
|
| 42 |
+
max_backoff_seconds: 900
|
| 43 |
start_date: '2024-01-01'
|
| 44 |
end_date: '2026-12-31'
|
| 45 |
|
docs/AGENTIC_SYSTEM_V1.md
ADDED
|
@@ -0,0 +1,130 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Algorithmic Trading
|
| 2 |
+
|
| 3 |
+
FinRL reinforcement-learning trading with Alpaca execution, plus Yahoo Finance OHLCV as the default public tape. Parallel LLC.
|
| 4 |
+
|
| 5 |
+
This is **research and paper-trading infrastructure**. Live capital requires a separate evaluation contract, feature-parity tests, and a rewritten execution path. Do not treat `paper_trading: false` as a promotion gate.
|
| 6 |
+
|
| 7 |
+
---
|
| 8 |
+
|
| 9 |
+
## 1. Title and Summary
|
| 10 |
+
|
| 11 |
+
**Algorithmic Trading**
|
| 12 |
+
Ingest OHLCV, compute indicators or train a FinRL policy, size orders under position and drawdown caps, route to paper or live Alpaca.
|
| 13 |
+
|
| 14 |
+
GitHub `main` is the FinRL / Docker / Streamlit tree plus algotrader 2.0. `dev` is the integration branch. Yahoo is the default `data_source.type`.
|
| 15 |
+
|
| 16 |
+
**Design themes**
|
| 17 |
+
|
| 18 |
+
* Four ingest paths: CSV replay, synthetic GBM, Alpaca REST, Yahoo (`yfinance>=1.0`)
|
| 19 |
+
* FinRL policies (PPO, A2C, DDPG, TD3) on a Gymnasium-style environment
|
| 20 |
+
* Alpaca for authenticated market data and order routing (paper by default)
|
| 21 |
+
* Yahoo for delayed public bars when no broker key is available
|
| 22 |
+
* Secrets from environment (`ALPACA_API_KEY`, `ALPACA_SECRET_KEY`), never committed
|
| 23 |
+
* Tests and Docker/CI as already present on this tree
|
| 24 |
+
|
| 25 |
+
---
|
| 26 |
+
|
| 27 |
+
## 2. Concepts and Methods
|
| 28 |
+
|
| 29 |
+
### Market data
|
| 30 |
+
|
| 31 |
+
| Source | When to use | Failure modes |
|
| 32 |
+
| ------ | ----------- | ------------- |
|
| 33 |
+
| **CSV** | Offline replay; default in `config.yaml` | Missing path or OHLCV columns → `None` |
|
| 34 |
+
| **Synthetic** | Unit tests and demos | GBM is not tradable edge |
|
| 35 |
+
| **Alpaca** | Authenticated bars and live/paper orders | Auth, feed, and rate-limit failures |
|
| 36 |
+
| **Yahoo** | Real Close without a broker account | Unofficial API, ~15 min delay, interval lookback caps (1m ≈ 7 days). Pin `yfinance>=1.0`; 0.2.x fails against the current chart API |
|
| 37 |
+
|
| 38 |
+
`load_data` dispatches on `data_source.type`. Existing `alpaca` / `csv` / `synthetic` branches are unchanged.
|
| 39 |
+
|
| 40 |
+
### Strategy and FinRL
|
| 41 |
+
|
| 42 |
+
* `StrategyAgent`: SMA, RSI, Bollinger, MACD on Close; teaching rule, not an alpha claim
|
| 43 |
+
* `FinRLAgent`: PPO / A2C / DDPG / TD3 via Stable-Baselines3; persist under `models/`
|
| 44 |
+
* `ExecutionAgent` / `AlpacaBroker`: paper simulation or Alpaca market/limit orders
|
| 45 |
+
|
| 46 |
+
Backtests in this repo are in-sample passes unless you add a purged walk-forward yourself. Leakage is the null hypothesis.
|
| 47 |
+
|
| 48 |
+
---
|
| 49 |
+
|
| 50 |
+
## 3. Stack
|
| 51 |
+
|
| 52 |
+
| Layer | Tools |
|
| 53 |
+
| ----- | ----- |
|
| 54 |
+
| Language | Python 3.11 (CI); 3.8+ stated for local |
|
| 55 |
+
| RL | FinRL / Stable-Baselines3, Gym/Gymnasium, PyTorch |
|
| 56 |
+
| Broker | alpaca-py |
|
| 57 |
+
| Market data | Alpaca REST; yfinance ≥ 1.0 (Yahoo) |
|
| 58 |
+
| Tabular | pandas, NumPy, scikit-learn |
|
| 59 |
+
| UI | Streamlit, Dash, Jupyter widgets |
|
| 60 |
+
| Deploy | Docker Compose, GitHub Actions |
|
| 61 |
+
| Tests | pytest, pytest-cov |
|
| 62 |
+
|
| 63 |
+
---
|
| 64 |
+
|
| 65 |
+
## 4. Structure
|
| 66 |
+
|
| 67 |
+
```
|
| 68 |
+
algorithmic_trading/
|
| 69 |
+
├── agentic_ai_system/ # ingest, strategy, FinRL, Alpaca, Yahoo
|
| 70 |
+
├── ui/ # Streamlit, Dash, Jupyter, WebSocket
|
| 71 |
+
├── tests/
|
| 72 |
+
├── models/ # trained artifacts (gitignored bodies)
|
| 73 |
+
├── data/ # generated CSV (gitignored)
|
| 74 |
+
├── scripts/ # Docker / deploy helpers
|
| 75 |
+
├── .github/workflows/ # CI/CD, release, backtesting
|
| 76 |
+
├── config.yaml
|
| 77 |
+
├── requirements.txt
|
| 78 |
+
├── Dockerfile
|
| 79 |
+
└── docker-compose*.yml
|
| 80 |
+
```
|
| 81 |
+
|
| 82 |
+
Branch policy: **`main`** (protected) and **`dev`** only. Do not re-enable Dependabot or the Monday `dependency-updates` workflow; those created extra branches.
|
| 83 |
+
|
| 84 |
+
---
|
| 85 |
+
|
| 86 |
+
## 5. Quick start
|
| 87 |
+
|
| 88 |
+
```bash
|
| 89 |
+
git clone https://github.com/ParallelLLC/algorithmic_trading.git
|
| 90 |
+
cd algorithmic_trading
|
| 91 |
+
python -m venv .venv && source .venv/bin/activate
|
| 92 |
+
pip install -r requirements.txt
|
| 93 |
+
cp .env.example .env # Alpaca keys if using alpaca ingest or orders
|
| 94 |
+
```
|
| 95 |
+
|
| 96 |
+
Default ingest is CSV. For Yahoo daily bars without a broker:
|
| 97 |
+
|
| 98 |
+
```yaml
|
| 99 |
+
data_source:
|
| 100 |
+
type: 'yahoo'
|
| 101 |
+
trading:
|
| 102 |
+
symbol: 'AAPL'
|
| 103 |
+
timeframe: '1d'
|
| 104 |
+
```
|
| 105 |
+
|
| 106 |
+
```bash
|
| 107 |
+
python demo.py
|
| 108 |
+
python -m agentic_ai_system.main --mode backtest --start-date 2024-01-01 --end-date 2024-12-31
|
| 109 |
+
pytest tests/ -q
|
| 110 |
+
```
|
| 111 |
+
|
| 112 |
+
UI launchers and Docker are documented in `UI_SETUP.md` and `DOCKER_HUB_SETUP.md`. Paper-trade before live. Yahoo is not a SIP tape.
|
| 113 |
+
|
| 114 |
+
---
|
| 115 |
+
|
| 116 |
+
## 6. Configuration (additive Yahoo keys)
|
| 117 |
+
|
| 118 |
+
| Key | Meaning |
|
| 119 |
+
| --- | ------- |
|
| 120 |
+
| `data_source.type` | `csv` \| `synthetic` \| `alpaca` \| `yahoo` |
|
| 121 |
+
| `yahoo.start_date` / `end_date` | Historical window; clamped per Yahoo interval limits |
|
| 122 |
+
| `yahoo.auto_adjust` | Passed to `yfinance` |
|
| 123 |
+
| `execution.broker_api` | `paper` \| `alpaca_paper` \| `alpaca_live` |
|
| 124 |
+
| `finrl.algorithm` | PPO, A2C, DDPG, TD3 |
|
| 125 |
+
|
| 126 |
+
---
|
| 127 |
+
|
| 128 |
+
**License:** Apache License 2.0
|
| 129 |
+
**Organization:** [Parallel LLC](https://github.com/ParallelLLC)
|
| 130 |
+
**Repository:** <https://github.com/ParallelLLC/algorithmic_trading>
|
requirements-space.txt
ADDED
|
@@ -0,0 +1,12 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Dependencies for the Hugging Face Space (app.py + the algotrader package).
|
| 2 |
+
#
|
| 3 |
+
# Deliberately minimal: a Space that takes ten minutes to build is a Space
|
| 4 |
+
# nobody waits for. The v1 agentic system's heavier stack (torch,
|
| 5 |
+
# stable-baselines3, dash, alpaca-py) lives in requirements.txt and is not
|
| 6 |
+
# needed to run the Lab.
|
| 7 |
+
gradio>=4.44,<7
|
| 8 |
+
numpy>=1.24
|
| 9 |
+
pandas>=2.0
|
| 10 |
+
scipy>=1.10
|
| 11 |
+
plotly>=5.18
|
| 12 |
+
yfinance>=0.2.40
|
scripts/deploy_hf_space.sh
ADDED
|
@@ -0,0 +1,65 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/usr/bin/env bash
|
| 2 |
+
#
|
| 3 |
+
# Assemble and push the Hugging Face Space.
|
| 4 |
+
#
|
| 5 |
+
# The Space gets only what it needs to run: app.py, the algotrader package, the
|
| 6 |
+
# Space card, and the minimal requirements file. The v1 agentic system and its
|
| 7 |
+
# heavy ML stack stay in this repo, so the Space builds in under a minute.
|
| 8 |
+
#
|
| 9 |
+
# Usage:
|
| 10 |
+
# HF_TOKEN=hf_xxx ./scripts/deploy_hf_space.sh <hf-username>/<space-name>
|
| 11 |
+
#
|
| 12 |
+
# Create the Space first at https://huggingface.co/new-space (SDK: Gradio).
|
| 13 |
+
|
| 14 |
+
set -euo pipefail
|
| 15 |
+
|
| 16 |
+
SPACE_ID="${1:-}"
|
| 17 |
+
if [[ -z "$SPACE_ID" ]]; then
|
| 18 |
+
echo "usage: HF_TOKEN=hf_xxx $0 <hf-username>/<space-name>" >&2
|
| 19 |
+
exit 2
|
| 20 |
+
fi
|
| 21 |
+
if [[ -z "${HF_TOKEN:-}" ]]; then
|
| 22 |
+
echo "error: HF_TOKEN is not set. Create a write token at https://huggingface.co/settings/tokens" >&2
|
| 23 |
+
exit 2
|
| 24 |
+
fi
|
| 25 |
+
|
| 26 |
+
REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
| 27 |
+
STAGING="$(mktemp -d)"
|
| 28 |
+
trap 'rm -rf "$STAGING"' EXIT
|
| 29 |
+
|
| 30 |
+
echo "==> Staging Space contents in $STAGING"
|
| 31 |
+
git clone --quiet "https://user:${HF_TOKEN}@huggingface.co/spaces/${SPACE_ID}" "$STAGING/space"
|
| 32 |
+
cd "$STAGING/space"
|
| 33 |
+
|
| 34 |
+
# Replace tracked content wholesale so deletions propagate, but keep .git.
|
| 35 |
+
find . -mindepth 1 -maxdepth 1 ! -name .git -exec rm -rf {} +
|
| 36 |
+
|
| 37 |
+
cp -r "$REPO_ROOT/algotrader" ./algotrader
|
| 38 |
+
cp "$REPO_ROOT/app.py" ./app.py
|
| 39 |
+
cp "$REPO_ROOT/LICENSE" ./LICENSE
|
| 40 |
+
cp "$REPO_ROOT/SPACE_README.md" ./README.md
|
| 41 |
+
cp "$REPO_ROOT/requirements-space.txt" ./requirements.txt
|
| 42 |
+
find ./algotrader -name '__pycache__' -type d -exec rm -rf {} + 2>/dev/null || true
|
| 43 |
+
|
| 44 |
+
# A small test suite ships too: reviewers who check whether the statistics are
|
| 45 |
+
# real are exactly the audience worth convincing.
|
| 46 |
+
mkdir -p tests
|
| 47 |
+
cp "$REPO_ROOT/tests/test_v2_engine.py" \
|
| 48 |
+
"$REPO_ROOT/tests/test_v2_validation.py" \
|
| 49 |
+
"$REPO_ROOT/tests/test_v2_strategies.py" \
|
| 50 |
+
"$REPO_ROOT/tests/test_v2_portfolio.py" tests/
|
| 51 |
+
|
| 52 |
+
echo "==> Files staged:"
|
| 53 |
+
find . -path ./.git -prune -o -type f -print | sed 's|^\./| |'
|
| 54 |
+
|
| 55 |
+
git add -A
|
| 56 |
+
if git diff --cached --quiet; then
|
| 57 |
+
echo "==> Space is already up to date; nothing to push."
|
| 58 |
+
exit 0
|
| 59 |
+
fi
|
| 60 |
+
|
| 61 |
+
git -c user.email="deploy@localhost" -c user.name="space-deploy" \
|
| 62 |
+
commit --quiet -m "Deploy algotrader $(python3 -c 'import sys; sys.path.insert(0, "'"$REPO_ROOT"'"); import algotrader; print(algotrader.__version__)')"
|
| 63 |
+
git push --quiet origin HEAD:main
|
| 64 |
+
|
| 65 |
+
echo "==> Pushed. Live at https://huggingface.co/spaces/${SPACE_ID}"
|
tests/test_data_ingestion.py
CHANGED
|
@@ -132,7 +132,7 @@ class TestDataIngestion:
|
|
| 132 |
|
| 133 |
assert isinstance(result, pd.DataFrame)
|
| 134 |
assert len(result) == len(sample_csv_data)
|
| 135 |
-
assert result['timestamp']
|
| 136 |
|
| 137 |
finally:
|
| 138 |
os.unlink(tmp_file.name)
|
|
@@ -289,7 +289,7 @@ class TestDataIngestion:
|
|
| 289 |
result = _load_csv_data(config)
|
| 290 |
|
| 291 |
# Check that timestamp is converted to datetime
|
| 292 |
-
assert result['timestamp']
|
| 293 |
|
| 294 |
finally:
|
| 295 |
os.unlink(tmp_file.name)
|
|
|
|
| 132 |
|
| 133 |
assert isinstance(result, pd.DataFrame)
|
| 134 |
assert len(result) == len(sample_csv_data)
|
| 135 |
+
assert pd.api.types.is_datetime64_any_dtype(result['timestamp'])
|
| 136 |
|
| 137 |
finally:
|
| 138 |
os.unlink(tmp_file.name)
|
|
|
|
| 289 |
result = _load_csv_data(config)
|
| 290 |
|
| 291 |
# Check that timestamp is converted to datetime
|
| 292 |
+
assert pd.api.types.is_datetime64_any_dtype(result['timestamp'])
|
| 293 |
|
| 294 |
finally:
|
| 295 |
os.unlink(tmp_file.name)
|
tests/test_synthetic_data_generator.py
CHANGED
|
@@ -56,8 +56,8 @@ class TestSyntheticDataGenerator:
|
|
| 56 |
assert col in df.columns
|
| 57 |
|
| 58 |
# Check data types
|
| 59 |
-
assert df['timestamp']
|
| 60 |
-
assert df['symbol']
|
| 61 |
assert df['open'].dtype in ['float64', 'float32']
|
| 62 |
assert df['high'].dtype in ['float64', 'float32']
|
| 63 |
assert df['low'].dtype in ['float64', 'float32']
|
|
|
|
| 56 |
assert col in df.columns
|
| 57 |
|
| 58 |
# Check data types
|
| 59 |
+
assert pd.api.types.is_datetime64_any_dtype(df['timestamp'])
|
| 60 |
+
assert pd.api.types.is_string_dtype(df['symbol'])
|
| 61 |
assert df['open'].dtype in ['float64', 'float32']
|
| 62 |
assert df['high'].dtype in ['float64', 'float32']
|
| 63 |
assert df['low'].dtype in ['float64', 'float32']
|
tests/test_v2_cli.py
ADDED
|
@@ -0,0 +1,89 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""CLI surface — argument handling and that each command actually completes."""
|
| 2 |
+
|
| 3 |
+
from __future__ import annotations
|
| 4 |
+
|
| 5 |
+
import json
|
| 6 |
+
|
| 7 |
+
import pytest
|
| 8 |
+
|
| 9 |
+
from algotrader.cli import _parse_params, main
|
| 10 |
+
|
| 11 |
+
|
| 12 |
+
class TestArgumentParsing:
|
| 13 |
+
def test_params_are_parsed_into_floats(self):
|
| 14 |
+
assert _parse_params(["fast=10", "slow=50.5"]) == {"fast": 10.0, "slow": 50.5}
|
| 15 |
+
|
| 16 |
+
def test_params_without_an_equals_sign_are_rejected(self):
|
| 17 |
+
with pytest.raises(SystemExit, match="name=value"):
|
| 18 |
+
_parse_params(["fast10"])
|
| 19 |
+
|
| 20 |
+
def test_no_params_is_an_empty_dict(self):
|
| 21 |
+
assert _parse_params(None) == {}
|
| 22 |
+
|
| 23 |
+
def test_an_unknown_strategy_is_rejected_at_parse_time(self):
|
| 24 |
+
with pytest.raises(SystemExit):
|
| 25 |
+
main(["lab", "--strategy", "nope"])
|
| 26 |
+
|
| 27 |
+
def test_a_missing_subcommand_is_rejected(self):
|
| 28 |
+
with pytest.raises(SystemExit):
|
| 29 |
+
main([])
|
| 30 |
+
|
| 31 |
+
|
| 32 |
+
class TestCommands:
|
| 33 |
+
"""All runs force the simulator so the suite never touches the network."""
|
| 34 |
+
|
| 35 |
+
def test_lab_prints_a_report(self, capsys):
|
| 36 |
+
code = main([
|
| 37 |
+
"lab", "--symbol", "SPY", "--source", "synthetic", "--strategy", "sma_cross",
|
| 38 |
+
"--start", "2018-01-01", "--end", "2023-01-01",
|
| 39 |
+
"--permutations", "10", "--folds", "2", "--quiet",
|
| 40 |
+
])
|
| 41 |
+
out = capsys.readouterr().out
|
| 42 |
+
assert code == 0
|
| 43 |
+
assert "REALITY SCORE" in out
|
| 44 |
+
assert "Deflated Sharpe" in out
|
| 45 |
+
|
| 46 |
+
def test_lab_json_output_is_machine_readable(self, capsys):
|
| 47 |
+
code = main([
|
| 48 |
+
"lab", "--source", "synthetic", "--start", "2018-01-01", "--end", "2023-01-01",
|
| 49 |
+
"--permutations", "10", "--folds", "2", "--quiet", "--json",
|
| 50 |
+
])
|
| 51 |
+
payload = json.loads(capsys.readouterr().out)
|
| 52 |
+
assert code == 0
|
| 53 |
+
assert 0.0 <= payload["verdict"]["score"] <= 100.0
|
| 54 |
+
assert payload["verdict"]["grade"] in {"A", "B", "C", "D", "F"}
|
| 55 |
+
assert payload["source"] == "synthetic"
|
| 56 |
+
|
| 57 |
+
def test_lab_honours_param_overrides(self, capsys):
|
| 58 |
+
main([
|
| 59 |
+
"lab", "--source", "synthetic", "--start", "2018-01-01", "--end", "2023-01-01",
|
| 60 |
+
"--strategy", "sma_cross", "--param", "fast=5", "--param", "slow=60",
|
| 61 |
+
"--permutations", "0", "--folds", "2", "--quiet", "--json",
|
| 62 |
+
])
|
| 63 |
+
payload = json.loads(capsys.readouterr().out)
|
| 64 |
+
assert payload["params"] == {"fast": 5, "slow": 60}
|
| 65 |
+
assert payload["p_value"] is None # permutations disabled
|
| 66 |
+
|
| 67 |
+
def test_arena_ranks_the_zoo(self, capsys):
|
| 68 |
+
code = main([
|
| 69 |
+
"arena", "--source", "synthetic", "--start", "2018-01-01", "--end", "2023-01-01",
|
| 70 |
+
"--permutations", "0", "--quiet",
|
| 71 |
+
])
|
| 72 |
+
out = capsys.readouterr().out
|
| 73 |
+
assert code == 0
|
| 74 |
+
assert "Buy & Hold" in out and "Coin Flip" in out
|
| 75 |
+
assert "Ranked by evidence" in out
|
| 76 |
+
|
| 77 |
+
def test_strategies_lists_the_registry(self, capsys):
|
| 78 |
+
assert main(["strategies"]) == 0
|
| 79 |
+
out = capsys.readouterr().out
|
| 80 |
+
assert "sma_cross" in out and "donchian_breakout" in out
|
| 81 |
+
|
| 82 |
+
def test_errors_are_reported_not_raised(self, capsys):
|
| 83 |
+
code = main([
|
| 84 |
+
"lab", "--source", "synthetic",
|
| 85 |
+
"--start", "2022-01-01", "--end", "2022-02-01", # far too short
|
| 86 |
+
"--permutations", "0", "--quiet",
|
| 87 |
+
])
|
| 88 |
+
assert code == 1
|
| 89 |
+
assert "error:" in capsys.readouterr().err
|
tests/test_v2_engine.py
ADDED
|
@@ -0,0 +1,158 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Engine correctness: alignment, costs, and the no-look-ahead guarantee."""
|
| 2 |
+
|
| 3 |
+
from __future__ import annotations
|
| 4 |
+
|
| 5 |
+
import numpy as np
|
| 6 |
+
import pandas as pd
|
| 7 |
+
import pytest
|
| 8 |
+
|
| 9 |
+
from algotrader.engine import bars_to_returns, run_backtest
|
| 10 |
+
from algotrader.metrics import compute_metrics, infer_periods_per_year, max_drawdown, sharpe_ratio
|
| 11 |
+
from algotrader.types import CostModel
|
| 12 |
+
|
| 13 |
+
|
| 14 |
+
def make_bars(n: int = 400, seed: int = 0) -> pd.DataFrame:
|
| 15 |
+
rng = np.random.default_rng(seed)
|
| 16 |
+
close = 100 * np.exp(np.cumsum(rng.normal(0.0003, 0.01, n)))
|
| 17 |
+
index = pd.date_range("2020-01-01", periods=n, freq="B")
|
| 18 |
+
return pd.DataFrame(
|
| 19 |
+
{
|
| 20 |
+
"open": close,
|
| 21 |
+
"high": close * 1.005,
|
| 22 |
+
"low": close * 0.995,
|
| 23 |
+
"close": close,
|
| 24 |
+
"volume": 1e6,
|
| 25 |
+
},
|
| 26 |
+
index=index,
|
| 27 |
+
)
|
| 28 |
+
|
| 29 |
+
|
| 30 |
+
class TestNoLookAhead:
|
| 31 |
+
"""The one property the whole project rests on."""
|
| 32 |
+
|
| 33 |
+
def test_signal_earns_the_following_bar_not_its_own(self):
|
| 34 |
+
df = make_bars()
|
| 35 |
+
asset_ret = bars_to_returns(df)
|
| 36 |
+
|
| 37 |
+
# A target that knows the *next* bar's direction must be perfect...
|
| 38 |
+
clairvoyant = np.sign(asset_ret.shift(-1)).fillna(0.0)
|
| 39 |
+
good = run_backtest(df, clairvoyant, costs=CostModel(0, 0, 0))
|
| 40 |
+
assert (good.returns.iloc[1:-1] >= -1e-12).all(), "perfect foresight should never lose"
|
| 41 |
+
|
| 42 |
+
# ...and a target that only knows the *current* bar must not be.
|
| 43 |
+
hindsight = np.sign(asset_ret).fillna(0.0)
|
| 44 |
+
meh = run_backtest(df, hindsight, costs=CostModel(0, 0, 0))
|
| 45 |
+
assert (meh.returns < 0).any(), "same-bar signal must not be risk-free"
|
| 46 |
+
assert good.sharpe > meh.sharpe
|
| 47 |
+
|
| 48 |
+
def test_position_is_target_shifted_by_lag(self):
|
| 49 |
+
df = make_bars(200)
|
| 50 |
+
target = pd.Series(np.linspace(-1, 1, len(df)), index=df.index)
|
| 51 |
+
for lag in (1, 2, 5):
|
| 52 |
+
result = run_backtest(df, target, lag=lag)
|
| 53 |
+
expected = target.shift(lag).fillna(0.0)
|
| 54 |
+
pd.testing.assert_series_equal(result.position, expected, check_names=False)
|
| 55 |
+
|
| 56 |
+
def test_lag_zero_is_rejected(self):
|
| 57 |
+
df = make_bars(120)
|
| 58 |
+
with pytest.raises(ValueError, match="lag"):
|
| 59 |
+
run_backtest(df, pd.Series(1.0, index=df.index), lag=0)
|
| 60 |
+
|
| 61 |
+
def test_future_bars_cannot_change_past_equity(self):
|
| 62 |
+
"""Truncating the data must not alter the equity curve before the cut."""
|
| 63 |
+
df = make_bars(400)
|
| 64 |
+
target = pd.Series(np.tile([1.0, -1.0], len(df) // 2), index=df.index)
|
| 65 |
+
|
| 66 |
+
full = run_backtest(df, target)
|
| 67 |
+
cut = run_backtest(df.iloc[:250], target.iloc[:250])
|
| 68 |
+
np.testing.assert_allclose(
|
| 69 |
+
full.equity.iloc[:250].to_numpy(), cut.equity.to_numpy(), rtol=1e-12
|
| 70 |
+
)
|
| 71 |
+
|
| 72 |
+
|
| 73 |
+
class TestCosts:
|
| 74 |
+
def test_buy_and_hold_pays_once(self):
|
| 75 |
+
df = make_bars(300)
|
| 76 |
+
target = pd.Series(1.0, index=df.index)
|
| 77 |
+
result = run_backtest(df, target, costs=CostModel(commission_bps=5, slippage_bps=5, short_borrow_bps=0))
|
| 78 |
+
# One entry at 10bps one-way, and no further turnover.
|
| 79 |
+
assert result.costs.sum() == pytest.approx(10 / 1e4, rel=1e-9)
|
| 80 |
+
assert int(result.metrics["n_trades"]) == 1
|
| 81 |
+
|
| 82 |
+
def test_flipping_every_bar_costs_more_than_holding(self):
|
| 83 |
+
df = make_bars(300)
|
| 84 |
+
costs = CostModel(commission_bps=5, slippage_bps=5, short_borrow_bps=0)
|
| 85 |
+
hold = run_backtest(df, pd.Series(1.0, index=df.index), costs=costs)
|
| 86 |
+
flip = run_backtest(df, pd.Series(np.tile([1.0, -1.0], 150), index=df.index), costs=costs)
|
| 87 |
+
assert flip.costs.sum() > 100 * hold.costs.sum()
|
| 88 |
+
|
| 89 |
+
def test_zero_costs_means_gross_equals_net(self):
|
| 90 |
+
df = make_bars(200)
|
| 91 |
+
target = pd.Series(np.tile([1.0, 0.0], 100), index=df.index)
|
| 92 |
+
result = run_backtest(df, target, costs=CostModel(0, 0, 0))
|
| 93 |
+
pd.testing.assert_series_equal(result.returns, result.gross_returns, check_names=False)
|
| 94 |
+
|
| 95 |
+
def test_short_borrow_is_charged_only_on_shorts(self):
|
| 96 |
+
df = make_bars(260)
|
| 97 |
+
costs = CostModel(commission_bps=0, slippage_bps=0, short_borrow_bps=365)
|
| 98 |
+
long_only = run_backtest(df, pd.Series(1.0, index=df.index), costs=costs)
|
| 99 |
+
short_only = run_backtest(df, pd.Series(-1.0, index=df.index), costs=costs)
|
| 100 |
+
assert long_only.costs.sum() == pytest.approx(0.0, abs=1e-12)
|
| 101 |
+
assert short_only.costs.sum() > 0
|
| 102 |
+
|
| 103 |
+
def test_higher_costs_never_improve_returns(self):
|
| 104 |
+
df = make_bars(300)
|
| 105 |
+
target = pd.Series(np.tile([1.0, -1.0], 150), index=df.index)
|
| 106 |
+
cheap = run_backtest(df, target, costs=CostModel(1, 1, 0))
|
| 107 |
+
dear = run_backtest(df, target, costs=CostModel(20, 20, 0))
|
| 108 |
+
assert dear.equity.iloc[-1] < cheap.equity.iloc[-1]
|
| 109 |
+
|
| 110 |
+
|
| 111 |
+
class TestConstraints:
|
| 112 |
+
def test_shorts_are_clipped_when_disallowed(self):
|
| 113 |
+
df = make_bars(150)
|
| 114 |
+
target = pd.Series(-1.0, index=df.index)
|
| 115 |
+
result = run_backtest(df, target, allow_short=False)
|
| 116 |
+
assert (result.target >= 0).all()
|
| 117 |
+
assert (result.position >= 0).all()
|
| 118 |
+
|
| 119 |
+
def test_leverage_is_clipped(self):
|
| 120 |
+
df = make_bars(150)
|
| 121 |
+
result = run_backtest(df, pd.Series(5.0, index=df.index), max_leverage=1.5)
|
| 122 |
+
assert result.target.max() == pytest.approx(1.5)
|
| 123 |
+
|
| 124 |
+
def test_empty_frame_is_rejected(self):
|
| 125 |
+
with pytest.raises(ValueError):
|
| 126 |
+
run_backtest(pd.DataFrame(columns=["open", "high", "low", "close", "volume"]), pd.Series(dtype=float))
|
| 127 |
+
|
| 128 |
+
|
| 129 |
+
class TestMetrics:
|
| 130 |
+
def test_sharpe_of_constant_returns_is_zero_not_infinite(self):
|
| 131 |
+
flat = pd.Series([0.001] * 100)
|
| 132 |
+
assert sharpe_ratio(flat, 252) == 0.0
|
| 133 |
+
|
| 134 |
+
def test_sharpe_scales_with_annualisation(self):
|
| 135 |
+
rng = np.random.default_rng(1)
|
| 136 |
+
returns = pd.Series(rng.normal(0.001, 0.01, 5000))
|
| 137 |
+
assert sharpe_ratio(returns, 252) == pytest.approx(sharpe_ratio(returns, 1) * np.sqrt(252))
|
| 138 |
+
|
| 139 |
+
def test_max_drawdown_matches_a_hand_worked_example(self):
|
| 140 |
+
equity = pd.Series([100.0, 120.0, 60.0, 90.0])
|
| 141 |
+
assert max_drawdown(equity) == pytest.approx(-0.5)
|
| 142 |
+
|
| 143 |
+
def test_buy_and_hold_metrics_match_the_price_series(self):
|
| 144 |
+
df = make_bars(500)
|
| 145 |
+
result = run_backtest(df, pd.Series(1.0, index=df.index), costs=CostModel(0, 0, 0))
|
| 146 |
+
expected = df["close"].iloc[-1] / df["close"].iloc[0] - 1.0
|
| 147 |
+
assert result.metrics["total_return"] == pytest.approx(expected, rel=1e-9)
|
| 148 |
+
|
| 149 |
+
def test_periodicity_inference(self):
|
| 150 |
+
daily = pd.date_range("2020-01-01", periods=300, freq="B")
|
| 151 |
+
assert 200 <= infer_periods_per_year(daily) <= 300
|
| 152 |
+
hourly = pd.date_range("2020-01-01", periods=300, freq="h")
|
| 153 |
+
assert infer_periods_per_year(hourly) > 1000
|
| 154 |
+
|
| 155 |
+
def test_metrics_survive_a_degenerate_series(self):
|
| 156 |
+
empty = pd.Series(dtype=float)
|
| 157 |
+
out = compute_metrics(empty, pd.Series(dtype=float))
|
| 158 |
+
assert "periods_per_year" in out
|
tests/test_v2_portfolio.py
ADDED
|
@@ -0,0 +1,377 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Multi-asset engine, panel, cross-sectional strategies and their null."""
|
| 2 |
+
|
| 3 |
+
from __future__ import annotations
|
| 4 |
+
|
| 5 |
+
import numpy as np
|
| 6 |
+
import pandas as pd
|
| 7 |
+
import pytest
|
| 8 |
+
|
| 9 |
+
from algotrader.attribution import build_style_factors, factor_attribution
|
| 10 |
+
from algotrader.cross_sectional import XS_REGISTRY, get_xs_strategy, scores_to_weights
|
| 11 |
+
from algotrader.data import simulate_ohlcv
|
| 12 |
+
from algotrader.engine import run_backtest
|
| 13 |
+
from algotrader.panel import Panel, load_panel
|
| 14 |
+
from algotrader.portfolio import (
|
| 15 |
+
normalise_weights,
|
| 16 |
+
rebalance_schedule,
|
| 17 |
+
run_portfolio_backtest,
|
| 18 |
+
)
|
| 19 |
+
from algotrader.types import CostModel
|
| 20 |
+
from algotrader.validation.cross_permutation import (
|
| 21 |
+
cross_sectional_permutation_test,
|
| 22 |
+
permute_within_dates,
|
| 23 |
+
)
|
| 24 |
+
|
| 25 |
+
XS_KEYS = sorted(XS_REGISTRY)
|
| 26 |
+
|
| 27 |
+
|
| 28 |
+
def make_panel(n_assets=8, n=1200, drift_sd=0.0, seed=1, common_vol=0.008, idio=0.012) -> Panel:
|
| 29 |
+
"""A synthetic universe. ``drift_sd`` controls genuine cross-sectional structure."""
|
| 30 |
+
rng = np.random.default_rng(seed)
|
| 31 |
+
index = pd.date_range("2016-01-01", periods=n, freq="B")
|
| 32 |
+
drift = rng.normal(0, drift_sd, n_assets)
|
| 33 |
+
returns = (
|
| 34 |
+
rng.normal(0.0002, common_vol, n)[:, None]
|
| 35 |
+
+ rng.normal(0, idio, (n, n_assets))
|
| 36 |
+
+ drift[None, :]
|
| 37 |
+
)
|
| 38 |
+
close = 100 * np.exp(np.cumsum(returns, axis=0))
|
| 39 |
+
columns = [f"A{i}" for i in range(n_assets)]
|
| 40 |
+
|
| 41 |
+
def frame(values):
|
| 42 |
+
return pd.DataFrame(values, index=index, columns=columns)
|
| 43 |
+
|
| 44 |
+
return Panel(
|
| 45 |
+
fields={
|
| 46 |
+
"open": frame(close),
|
| 47 |
+
"high": frame(close * 1.004),
|
| 48 |
+
"low": frame(close * 0.996),
|
| 49 |
+
"close": frame(close),
|
| 50 |
+
"volume": frame(np.full_like(close, 1e6)),
|
| 51 |
+
},
|
| 52 |
+
sources={c: "synthetic" for c in columns},
|
| 53 |
+
)
|
| 54 |
+
|
| 55 |
+
|
| 56 |
+
@pytest.fixture(scope="module")
|
| 57 |
+
def panel() -> Panel:
|
| 58 |
+
return make_panel()
|
| 59 |
+
|
| 60 |
+
|
| 61 |
+
class TestPanel:
|
| 62 |
+
def test_fields_are_aligned(self, panel):
|
| 63 |
+
assert panel.shape == (1200, 8)
|
| 64 |
+
for name in ("open", "high", "low", "close", "volume"):
|
| 65 |
+
assert panel.fields[name].index.equals(panel.index)
|
| 66 |
+
assert list(panel.fields[name].columns) == panel.symbols
|
| 67 |
+
|
| 68 |
+
def test_misaligned_fields_are_rejected(self, panel):
|
| 69 |
+
broken = dict(panel.fields)
|
| 70 |
+
broken["high"] = broken["high"].iloc[:-5]
|
| 71 |
+
with pytest.raises(ValueError, match="not aligned"):
|
| 72 |
+
Panel(fields=broken)
|
| 73 |
+
|
| 74 |
+
def test_missing_field_is_rejected(self, panel):
|
| 75 |
+
with pytest.raises(ValueError, match="missing field"):
|
| 76 |
+
Panel(fields={"close": panel.close})
|
| 77 |
+
|
| 78 |
+
def test_missing_data_is_not_forward_filled(self):
|
| 79 |
+
"""Filling a gap invents liquidity that never existed."""
|
| 80 |
+
df = simulate_ohlcv("AAPL", "2018-01-01", "2022-01-01")
|
| 81 |
+
gapped = df.drop(df.index[100:140])
|
| 82 |
+
built = Panel.from_frames({"AAPL": gapped, "MSFT": df})
|
| 83 |
+
assert built.close["AAPL"].isna().sum() == 40
|
| 84 |
+
|
| 85 |
+
def test_tradable_requires_two_consecutive_prices(self, panel):
|
| 86 |
+
assert not panel.tradable().iloc[0].any()
|
| 87 |
+
assert panel.tradable().iloc[1:].all().all()
|
| 88 |
+
|
| 89 |
+
def test_returns_are_masked_where_untradable(self, panel):
|
| 90 |
+
assert panel.returns().iloc[0].isna().all()
|
| 91 |
+
|
| 92 |
+
def test_delisting_is_detected(self):
|
| 93 |
+
df = simulate_ohlcv("AAPL", "2016-01-01", "2022-01-01")
|
| 94 |
+
dead = df.iloc[: len(df) // 2]
|
| 95 |
+
built = Panel.from_frames({"ALIVE": df, "DEAD": dead})
|
| 96 |
+
report = built.survivorship()
|
| 97 |
+
assert report.n_delisted == 1
|
| 98 |
+
assert "DEAD" in report.delisted_symbols
|
| 99 |
+
assert not report.biased
|
| 100 |
+
|
| 101 |
+
def test_all_survivors_is_flagged_as_biased(self, panel):
|
| 102 |
+
report = panel.survivorship()
|
| 103 |
+
assert report.survival_rate == 1.0
|
| 104 |
+
assert report.biased
|
| 105 |
+
assert "upper bound" in report.note
|
| 106 |
+
|
| 107 |
+
def test_load_panel_skips_symbols_without_history(self):
|
| 108 |
+
built = load_panel(["SPY", "AAPL"], "2018-01-01", "2022-01-01", source="synthetic")
|
| 109 |
+
assert set(built.symbols) == {"SPY", "AAPL"}
|
| 110 |
+
assert all(s == "synthetic" for s in built.sources.values())
|
| 111 |
+
|
| 112 |
+
def test_select_and_slice(self, panel):
|
| 113 |
+
subset = panel.select(["A0", "A1"])
|
| 114 |
+
assert subset.symbols == ["A0", "A1"]
|
| 115 |
+
sliced = panel.slice(panel.index[10], panel.index[50])
|
| 116 |
+
assert len(sliced) == 41
|
| 117 |
+
|
| 118 |
+
|
| 119 |
+
class TestPortfolioEngine:
|
| 120 |
+
def test_matches_single_asset_engine_exactly(self):
|
| 121 |
+
"""One column through the matrix engine must equal the fast path."""
|
| 122 |
+
df = simulate_ohlcv("SPY", "2018-01-01", "2023-01-01")
|
| 123 |
+
single = Panel.from_frames({"SPY": df})
|
| 124 |
+
costs = CostModel(1, 2, 50)
|
| 125 |
+
n = len(df)
|
| 126 |
+
cases = {
|
| 127 |
+
"buy_and_hold": np.ones(n),
|
| 128 |
+
"flip": np.tile([1.0, -1.0], n // 2 + 1)[:n],
|
| 129 |
+
"long_flat": np.tile([1.0, 0.0], n // 2 + 1)[:n],
|
| 130 |
+
"fractional": np.full(n, 0.5),
|
| 131 |
+
}
|
| 132 |
+
for name, target in cases.items():
|
| 133 |
+
portfolio = run_portfolio_backtest(
|
| 134 |
+
single, pd.DataFrame({"SPY": target}, index=df.index), costs=costs
|
| 135 |
+
)
|
| 136 |
+
direct = run_backtest(df, pd.Series(target, index=df.index), costs=costs)
|
| 137 |
+
np.testing.assert_allclose(
|
| 138 |
+
portfolio.returns.to_numpy(), direct.returns.to_numpy(),
|
| 139 |
+
atol=1e-12, err_msg=f"mismatch for {name}",
|
| 140 |
+
)
|
| 141 |
+
|
| 142 |
+
def test_holding_still_costs_nothing_but_drift_does(self, panel):
|
| 143 |
+
"""A full-notional long needs no rebalancing; a short does."""
|
| 144 |
+
costs = CostModel(10, 10, 0)
|
| 145 |
+
long_only = pd.DataFrame(1.0 / len(panel.symbols), index=panel.index, columns=panel.symbols)
|
| 146 |
+
result = run_portfolio_backtest(panel, long_only, costs=costs, gross_leverage=1.0)
|
| 147 |
+
# Equal-weight across many names still drifts apart, so turnover > 0...
|
| 148 |
+
assert result.metrics["turnover_ann"] > 0
|
| 149 |
+
# ...but a single full-notional position does not.
|
| 150 |
+
one = pd.DataFrame(0.0, index=panel.index, columns=panel.symbols)
|
| 151 |
+
one["A0"] = 1.0
|
| 152 |
+
assert run_portfolio_backtest(panel, one, costs=costs).costs.sum() == pytest.approx(
|
| 153 |
+
20 / 1e4, rel=1e-6
|
| 154 |
+
)
|
| 155 |
+
|
| 156 |
+
def test_rebalancing_less_often_lowers_turnover(self, panel):
|
| 157 |
+
weights = get_xs_strategy("equal_weight").generate(panel)
|
| 158 |
+
turnovers = []
|
| 159 |
+
for frequency in ("D", "M", "Q"):
|
| 160 |
+
result = run_portfolio_backtest(
|
| 161 |
+
panel, weights, costs=CostModel(1, 2, 0),
|
| 162 |
+
rebalance_on=rebalance_schedule(panel.index, frequency),
|
| 163 |
+
)
|
| 164 |
+
turnovers.append(result.metrics["turnover_ann"])
|
| 165 |
+
assert turnovers[0] > turnovers[1] > turnovers[2]
|
| 166 |
+
|
| 167 |
+
def test_gross_leverage_is_capped(self, panel):
|
| 168 |
+
weights = pd.DataFrame(1.0, index=panel.index, columns=panel.symbols)
|
| 169 |
+
result = run_portfolio_backtest(panel, weights, gross_leverage=1.0)
|
| 170 |
+
assert result.weights.abs().sum(axis=1).max() <= 1.0 + 1e-9
|
| 171 |
+
|
| 172 |
+
def test_leverage_is_scaled_down_never_up(self, panel):
|
| 173 |
+
small = pd.DataFrame(0.01, index=panel.index, columns=panel.symbols)
|
| 174 |
+
result = run_portfolio_backtest(panel, small, gross_leverage=1.0)
|
| 175 |
+
assert result.weights.abs().sum(axis=1).max() < 0.2
|
| 176 |
+
|
| 177 |
+
def test_per_name_cap_is_applied(self, panel):
|
| 178 |
+
weights = pd.DataFrame(0.0, index=panel.index, columns=panel.symbols)
|
| 179 |
+
weights["A0"] = 1.0
|
| 180 |
+
result = run_portfolio_backtest(panel, weights, max_weight=0.1)
|
| 181 |
+
assert result.weights.abs().max().max() <= 0.1 + 1e-9
|
| 182 |
+
|
| 183 |
+
def test_shorts_are_blocked_when_disallowed(self, panel):
|
| 184 |
+
weights = pd.DataFrame(-0.1, index=panel.index, columns=panel.symbols)
|
| 185 |
+
result = run_portfolio_backtest(panel, weights, allow_short=False)
|
| 186 |
+
assert (result.weights >= 0).all().all()
|
| 187 |
+
|
| 188 |
+
def test_delisted_names_cannot_be_held(self):
|
| 189 |
+
df = simulate_ohlcv("AAPL", "2016-01-01", "2022-01-01")
|
| 190 |
+
built = Panel.from_frames({"ALIVE": df, "DEAD": df.iloc[: len(df) // 2]})
|
| 191 |
+
weights = pd.DataFrame(0.5, index=built.index, columns=built.symbols)
|
| 192 |
+
result = run_portfolio_backtest(built, weights)
|
| 193 |
+
after_death = built.close["DEAD"].last_valid_index()
|
| 194 |
+
assert result.held.loc[result.held.index > after_death, "DEAD"].abs().max() == 0.0
|
| 195 |
+
|
| 196 |
+
def test_lag_zero_is_rejected(self, panel):
|
| 197 |
+
with pytest.raises(ValueError, match="lag"):
|
| 198 |
+
run_portfolio_backtest(panel, pd.DataFrame(0.1, index=panel.index, columns=panel.symbols), lag=0)
|
| 199 |
+
|
| 200 |
+
def test_future_bars_cannot_change_past_equity(self, panel):
|
| 201 |
+
weights = get_xs_strategy("xs_momentum").generate(panel)
|
| 202 |
+
full = run_portfolio_backtest(panel, weights)
|
| 203 |
+
cut = run_portfolio_backtest(
|
| 204 |
+
panel.slice(panel.index[0], panel.index[799]), weights.iloc[:800]
|
| 205 |
+
)
|
| 206 |
+
np.testing.assert_allclose(
|
| 207 |
+
full.equity.iloc[:800].to_numpy(), cut.equity.to_numpy(), rtol=1e-10
|
| 208 |
+
)
|
| 209 |
+
|
| 210 |
+
def test_attribution_sums_to_gross_return(self, panel):
|
| 211 |
+
weights = get_xs_strategy("equal_weight").generate(panel)
|
| 212 |
+
result = run_portfolio_backtest(panel, weights, costs=CostModel(0, 0, 0))
|
| 213 |
+
assert result.attribution().sum() == pytest.approx(result.gross_returns.sum(), rel=1e-9)
|
| 214 |
+
|
| 215 |
+
def test_portfolio_metrics_are_reported(self, panel):
|
| 216 |
+
result = run_portfolio_backtest(panel, get_xs_strategy("xs_momentum").generate(panel))
|
| 217 |
+
for key in ("gross_exposure", "net_exposure", "avg_positions", "concentration_hhi"):
|
| 218 |
+
assert key in result.metrics
|
| 219 |
+
|
| 220 |
+
|
| 221 |
+
class TestCrossSectionalStrategies:
|
| 222 |
+
@pytest.mark.parametrize("key", XS_KEYS)
|
| 223 |
+
def test_weights_are_well_formed(self, panel, key):
|
| 224 |
+
weights = get_xs_strategy(key).generate(panel)
|
| 225 |
+
assert weights.shape == panel.shape
|
| 226 |
+
assert weights.notna().all().all()
|
| 227 |
+
assert weights.abs().sum(axis=1).max() <= 1.0 + 1e-9
|
| 228 |
+
|
| 229 |
+
@pytest.mark.parametrize("key", XS_KEYS)
|
| 230 |
+
def test_weights_are_causal(self, panel, key):
|
| 231 |
+
cut = 700
|
| 232 |
+
full = get_xs_strategy(key).generate(panel).iloc[:cut]
|
| 233 |
+
partial = get_xs_strategy(key).generate(panel.slice(panel.index[0], panel.index[cut - 1]))
|
| 234 |
+
pd.testing.assert_frame_equal(full, partial, rtol=1e-9)
|
| 235 |
+
|
| 236 |
+
def test_long_short_books_are_dollar_neutral(self, panel):
|
| 237 |
+
weights = get_xs_strategy("xs_momentum").generate(panel)
|
| 238 |
+
active = weights[weights.abs().sum(axis=1) > 1e-9]
|
| 239 |
+
assert active.sum(axis=1).abs().max() < 1e-9
|
| 240 |
+
|
| 241 |
+
def test_long_only_uses_the_whole_book(self):
|
| 242 |
+
scores = pd.DataFrame(np.arange(40).reshape(4, 10).astype(float))
|
| 243 |
+
weights = scores_to_weights(scores, long_frac=0.3, long_only=True)
|
| 244 |
+
assert weights.sum(axis=1).round(9).eq(1.0).all()
|
| 245 |
+
assert (weights >= 0).all().all()
|
| 246 |
+
|
| 247 |
+
def test_thin_cross_sections_are_skipped(self):
|
| 248 |
+
scores = pd.DataFrame(np.random.default_rng(0).normal(size=(10, 3)))
|
| 249 |
+
assert scores_to_weights(scores).abs().sum(axis=1).max() == 0.0
|
| 250 |
+
|
| 251 |
+
def test_equal_weight_holds_everything(self, panel):
|
| 252 |
+
weights = get_xs_strategy("equal_weight").generate(panel)
|
| 253 |
+
assert (weights.iloc[-1] > 0).all()
|
| 254 |
+
|
| 255 |
+
def test_unknown_strategy_names_alternatives(self):
|
| 256 |
+
with pytest.raises(KeyError, match="Available"):
|
| 257 |
+
get_xs_strategy("nope")
|
| 258 |
+
|
| 259 |
+
|
| 260 |
+
class TestCrossSectionalNull:
|
| 261 |
+
def test_permutation_preserves_the_shape_of_the_book(self, panel):
|
| 262 |
+
weights = get_xs_strategy("xs_momentum").generate(panel)
|
| 263 |
+
investable = panel.close.notna()
|
| 264 |
+
shuffled = permute_within_dates(weights, investable, np.random.default_rng(0))
|
| 265 |
+
|
| 266 |
+
np.testing.assert_allclose(
|
| 267 |
+
np.sort(weights.to_numpy(), axis=1), np.sort(shuffled.to_numpy(), axis=1), atol=1e-12
|
| 268 |
+
)
|
| 269 |
+
np.testing.assert_allclose(
|
| 270 |
+
weights.abs().sum(axis=1).to_numpy(), shuffled.abs().sum(axis=1).to_numpy(), atol=1e-12
|
| 271 |
+
)
|
| 272 |
+
np.testing.assert_allclose(
|
| 273 |
+
weights.sum(axis=1).to_numpy(), shuffled.sum(axis=1).to_numpy(), atol=1e-12
|
| 274 |
+
)
|
| 275 |
+
|
| 276 |
+
def test_permutation_actually_reassigns(self, panel):
|
| 277 |
+
weights = get_xs_strategy("xs_momentum").generate(panel)
|
| 278 |
+
shuffled = permute_within_dates(weights, panel.close.notna(), np.random.default_rng(0))
|
| 279 |
+
assert not np.allclose(weights.to_numpy(), shuffled.to_numpy())
|
| 280 |
+
|
| 281 |
+
def test_non_investable_names_stay_empty(self):
|
| 282 |
+
df = simulate_ohlcv("AAPL", "2016-01-01", "2022-01-01")
|
| 283 |
+
built = Panel.from_frames({"ALIVE": df, "DEAD": df.iloc[: len(df) // 2]})
|
| 284 |
+
weights = pd.DataFrame(0.5, index=built.index, columns=built.symbols)
|
| 285 |
+
shuffled = permute_within_dates(weights, built.close.notna(), np.random.default_rng(1))
|
| 286 |
+
after_death = built.close["DEAD"].last_valid_index()
|
| 287 |
+
assert shuffled.loc[shuffled.index > after_death, "DEAD"].abs().max() == 0.0
|
| 288 |
+
|
| 289 |
+
def test_real_cross_sectional_skill_is_detected(self):
|
| 290 |
+
strong = make_panel(drift_sd=0.003, seed=11)
|
| 291 |
+
weights = get_xs_strategy("xs_momentum").generate(strong)
|
| 292 |
+
result = cross_sectional_permutation_test(
|
| 293 |
+
strong, weights, n_permutations=100,
|
| 294 |
+
rebalance_on=rebalance_schedule(strong.index, "M"), seed=2,
|
| 295 |
+
)
|
| 296 |
+
assert result.observed > result.null_mean
|
| 297 |
+
assert result.p_value < 0.05
|
| 298 |
+
|
| 299 |
+
def test_no_structure_is_not_significant(self):
|
| 300 |
+
flat = make_panel(drift_sd=0.0, seed=12)
|
| 301 |
+
weights = get_xs_strategy("xs_momentum").generate(flat)
|
| 302 |
+
result = cross_sectional_permutation_test(
|
| 303 |
+
flat, weights, n_permutations=100,
|
| 304 |
+
rebalance_on=rebalance_schedule(flat.index, "M"), seed=4,
|
| 305 |
+
)
|
| 306 |
+
assert result.p_value > 0.05
|
| 307 |
+
|
| 308 |
+
def test_p_value_can_never_be_zero(self, panel):
|
| 309 |
+
weights = get_xs_strategy("xs_momentum").generate(panel)
|
| 310 |
+
result = cross_sectional_permutation_test(panel, weights, n_permutations=20, seed=0)
|
| 311 |
+
assert result.p_value >= 1 / 21
|
| 312 |
+
|
| 313 |
+
|
| 314 |
+
class TestAttribution:
|
| 315 |
+
def test_equal_weight_is_explained_by_the_market(self, panel):
|
| 316 |
+
factors = build_style_factors(panel)
|
| 317 |
+
weights = get_xs_strategy("equal_weight").generate(panel)
|
| 318 |
+
result = run_portfolio_backtest(panel, weights)
|
| 319 |
+
report = factor_attribution(result.returns, factors)
|
| 320 |
+
|
| 321 |
+
assert report["available"]
|
| 322 |
+
assert report["r_squared"] > 0.85
|
| 323 |
+
assert report["dominant_factor"] == "market"
|
| 324 |
+
assert not report["alpha_significant"]
|
| 325 |
+
assert "more cheaply" in report["note"]
|
| 326 |
+
|
| 327 |
+
def test_a_factor_regressed_on_itself_has_no_alpha(self, panel):
|
| 328 |
+
factors = build_style_factors(panel)
|
| 329 |
+
report = factor_attribution(factors["market"], factors)
|
| 330 |
+
assert abs(report["alpha_annual"]) < 0.05
|
| 331 |
+
assert report["r_squared"] > 0.99
|
| 332 |
+
|
| 333 |
+
def test_pure_noise_has_no_alpha_and_no_fit(self, panel):
|
| 334 |
+
rng = np.random.default_rng(5)
|
| 335 |
+
noise = pd.Series(rng.normal(0, 0.01, len(panel)), index=panel.index)
|
| 336 |
+
report = factor_attribution(noise, build_style_factors(panel))
|
| 337 |
+
assert not report["alpha_significant"]
|
| 338 |
+
assert report["r_squared"] < 0.2
|
| 339 |
+
|
| 340 |
+
def test_short_series_degrade_gracefully(self, panel):
|
| 341 |
+
factors = build_style_factors(panel)
|
| 342 |
+
report = factor_attribution(factors["market"].iloc[:20], factors)
|
| 343 |
+
assert not report["available"]
|
| 344 |
+
assert report["note"]
|
| 345 |
+
|
| 346 |
+
|
| 347 |
+
class TestPortfolioLab:
|
| 348 |
+
def test_full_pipeline(self):
|
| 349 |
+
from algotrader.portfolio_lab import PortfolioLabConfig, run_portfolio_lab
|
| 350 |
+
|
| 351 |
+
report = run_portfolio_lab(
|
| 352 |
+
PortfolioLabConfig(
|
| 353 |
+
symbols=["SPY", "QQQ", "AAPL", "MSFT", "NVDA", "GLD"],
|
| 354 |
+
start="2017-01-01", end="2023-01-01", source="synthetic",
|
| 355 |
+
strategy="xs_momentum", n_permutations=20, wf_folds=2, grid_limit=6,
|
| 356 |
+
)
|
| 357 |
+
)
|
| 358 |
+
assert 0.0 <= report.verdict["score"] <= 100.0
|
| 359 |
+
assert report.verdict["grade"] in {"A", "B", "C", "D", "F"}
|
| 360 |
+
assert 0 < report.permutation.p_value <= 1
|
| 361 |
+
assert report.attribution["available"]
|
| 362 |
+
assert report.survivorship.n_symbols == 6
|
| 363 |
+
assert report.cost_stress["sharpe_3x"] <= report.cost_stress["sharpe_1x"] + 1e-9
|
| 364 |
+
|
| 365 |
+
def test_survivorship_bias_reaches_the_verdict(self):
|
| 366 |
+
from algotrader.portfolio_lab import PortfolioLabConfig, run_portfolio_lab
|
| 367 |
+
|
| 368 |
+
report = run_portfolio_lab(
|
| 369 |
+
PortfolioLabConfig(
|
| 370 |
+
symbols=["SPY", "QQQ", "AAPL", "MSFT", "NVDA", "GLD"],
|
| 371 |
+
start="2015-01-01", end="2023-01-01", source="synthetic",
|
| 372 |
+
strategy="equal_weight", n_permutations=0, wf_folds=2, grid_limit=3,
|
| 373 |
+
)
|
| 374 |
+
)
|
| 375 |
+
assert report.survivorship.biased
|
| 376 |
+
assert any("still trading" in f for f in report.verdict["flags"])
|
| 377 |
+
assert report.verdict["score"] <= 60
|
tests/test_v2_strategies.py
ADDED
|
@@ -0,0 +1,234 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Strategy zoo, data loading, and the end-to-end Lab pipeline."""
|
| 2 |
+
|
| 3 |
+
from __future__ import annotations
|
| 4 |
+
|
| 5 |
+
import numpy as np
|
| 6 |
+
import pandas as pd
|
| 7 |
+
import pytest
|
| 8 |
+
|
| 9 |
+
from algotrader import LabConfig, run_lab
|
| 10 |
+
from algotrader.data import _normalise, load_ohlcv, simulate_ohlcv
|
| 11 |
+
from algotrader.indicators import atr, bollinger, donchian, ema, macd, rsi, sma
|
| 12 |
+
from algotrader.lab import run_arena
|
| 13 |
+
from algotrader.strategies import REGISTRY, get_strategy, list_strategies
|
| 14 |
+
from algotrader.types import MarketData
|
| 15 |
+
from algotrader.verdict import reality_score
|
| 16 |
+
|
| 17 |
+
ALL_KEYS = sorted(REGISTRY)
|
| 18 |
+
|
| 19 |
+
|
| 20 |
+
@pytest.fixture(scope="module")
|
| 21 |
+
def bars() -> pd.DataFrame:
|
| 22 |
+
return simulate_ohlcv("AAPL", "2016-01-01", "2023-01-01")
|
| 23 |
+
|
| 24 |
+
|
| 25 |
+
class TestIndicatorsAreCausal:
|
| 26 |
+
"""An indicator at bar t must not move when bars after t arrive."""
|
| 27 |
+
|
| 28 |
+
@pytest.mark.parametrize(
|
| 29 |
+
"fn",
|
| 30 |
+
[
|
| 31 |
+
lambda d: sma(d["close"], 20),
|
| 32 |
+
lambda d: ema(d["close"], 12),
|
| 33 |
+
lambda d: rsi(d["close"], 14),
|
| 34 |
+
lambda d: macd(d["close"])[0],
|
| 35 |
+
lambda d: bollinger(d["close"])[2],
|
| 36 |
+
lambda d: atr(d, 14),
|
| 37 |
+
lambda d: donchian(d, 20)[1],
|
| 38 |
+
],
|
| 39 |
+
)
|
| 40 |
+
def test_prefix_is_stable(self, bars, fn):
|
| 41 |
+
cut = 800
|
| 42 |
+
full = fn(bars).iloc[:cut]
|
| 43 |
+
partial = fn(bars.iloc[:cut])
|
| 44 |
+
pd.testing.assert_series_equal(full, partial, check_names=False, rtol=1e-9)
|
| 45 |
+
|
| 46 |
+
def test_rsi_stays_in_range(self, bars):
|
| 47 |
+
values = rsi(bars["close"], 14).dropna()
|
| 48 |
+
assert values.between(0, 100).all()
|
| 49 |
+
|
| 50 |
+
def test_donchian_excludes_the_current_bar(self, bars):
|
| 51 |
+
_, upper = donchian(bars, 20)
|
| 52 |
+
# A breakout must be possible: the channel cannot already contain today.
|
| 53 |
+
assert (bars["high"] > upper).any()
|
| 54 |
+
|
| 55 |
+
|
| 56 |
+
class TestStrategies:
|
| 57 |
+
@pytest.mark.parametrize("key", ALL_KEYS)
|
| 58 |
+
def test_output_is_well_formed(self, bars, key):
|
| 59 |
+
target = get_strategy(key).generate(bars)
|
| 60 |
+
assert target.index.equals(bars.index)
|
| 61 |
+
assert target.notna().all()
|
| 62 |
+
assert target.between(-1.0, 1.0).all()
|
| 63 |
+
|
| 64 |
+
@pytest.mark.parametrize("key", ALL_KEYS)
|
| 65 |
+
def test_signals_are_causal(self, bars, key):
|
| 66 |
+
cut = 900
|
| 67 |
+
full = get_strategy(key).generate(bars).iloc[:cut]
|
| 68 |
+
partial = get_strategy(key).generate(bars.iloc[:cut])
|
| 69 |
+
pd.testing.assert_series_equal(full, partial, check_names=False, rtol=1e-9)
|
| 70 |
+
|
| 71 |
+
@pytest.mark.parametrize("key", ALL_KEYS)
|
| 72 |
+
def test_grid_is_non_empty_and_valid(self, key):
|
| 73 |
+
strategy = get_strategy(key)
|
| 74 |
+
grid = strategy.grid(limit=40)
|
| 75 |
+
assert 1 <= len(grid) <= 40
|
| 76 |
+
for combo in grid:
|
| 77 |
+
assert set(combo) <= {p.name for p in strategy.params}
|
| 78 |
+
if "fast" in combo and "slow" in combo:
|
| 79 |
+
assert combo["fast"] < combo["slow"]
|
| 80 |
+
|
| 81 |
+
def test_unknown_strategy_names_the_alternatives(self):
|
| 82 |
+
with pytest.raises(KeyError, match="Available"):
|
| 83 |
+
get_strategy("does_not_exist")
|
| 84 |
+
|
| 85 |
+
def test_clean_ignores_unknown_params_and_casts_types(self):
|
| 86 |
+
strategy = get_strategy("sma_cross")
|
| 87 |
+
cleaned = strategy.clean({"fast": 15.7, "nonsense": 1})
|
| 88 |
+
assert cleaned == {"fast": 15, "slow": 100}
|
| 89 |
+
|
| 90 |
+
def test_buy_and_hold_is_always_fully_invested(self, bars):
|
| 91 |
+
assert (get_strategy("buy_and_hold").generate(bars) == 1.0).all()
|
| 92 |
+
|
| 93 |
+
def test_list_strategies_honours_exclusions(self):
|
| 94 |
+
keys = {s.key for s in list_strategies(exclude=["coin_flip"])}
|
| 95 |
+
assert "coin_flip" not in keys and "sma_cross" in keys
|
| 96 |
+
|
| 97 |
+
|
| 98 |
+
class TestData:
|
| 99 |
+
def test_simulation_is_deterministic_per_symbol(self):
|
| 100 |
+
a = simulate_ohlcv("NVDA", "2018-01-01", "2022-01-01")
|
| 101 |
+
b = simulate_ohlcv("NVDA", "2018-01-01", "2022-01-01")
|
| 102 |
+
pd.testing.assert_frame_equal(a, b)
|
| 103 |
+
|
| 104 |
+
def test_different_symbols_simulate_differently(self):
|
| 105 |
+
a = simulate_ohlcv("NVDA", "2018-01-01", "2022-01-01")
|
| 106 |
+
b = simulate_ohlcv("TSLA", "2018-01-01", "2022-01-01")
|
| 107 |
+
assert not np.allclose(a["close"].to_numpy(), b["close"].to_numpy())
|
| 108 |
+
|
| 109 |
+
def test_simulated_bars_are_internally_consistent(self):
|
| 110 |
+
df = simulate_ohlcv("SPY", "2015-01-01", "2023-01-01")
|
| 111 |
+
assert (df["high"] >= df[["open", "close"]].max(axis=1) - 1e-9).all()
|
| 112 |
+
assert (df["low"] <= df[["open", "close"]].min(axis=1) + 1e-9).all()
|
| 113 |
+
assert (df["close"] > 0).all()
|
| 114 |
+
assert df.index.is_monotonic_increasing
|
| 115 |
+
|
| 116 |
+
def test_simulation_has_fat_tails_and_vol_clustering(self):
|
| 117 |
+
"""Naive GBM flatters strategies; the simulator must be harder than that."""
|
| 118 |
+
returns = simulate_ohlcv("SPY", "2005-01-01", "2023-01-01")["close"].pct_change().dropna()
|
| 119 |
+
assert returns.kurtosis() > 1.0
|
| 120 |
+
assert returns.abs().autocorr(1) > 0.05
|
| 121 |
+
|
| 122 |
+
def test_yahoo_source_does_not_silently_simulate(self, monkeypatch):
|
| 123 |
+
monkeypatch.setattr("algotrader.data._download", lambda *a, **k: None)
|
| 124 |
+
with pytest.raises(RuntimeError, match="Yahoo returned no usable bars"):
|
| 125 |
+
load_ohlcv("SPY", "2018-01-01", "2022-01-01", source="yahoo")
|
| 126 |
+
|
| 127 |
+
def test_offline_load_falls_back_and_says_so(self, monkeypatch):
|
| 128 |
+
monkeypatch.setattr("algotrader.data._download", lambda *a, **k: None)
|
| 129 |
+
monkeypatch.setattr("algotrader.data._read_cache", lambda *a, **k: None)
|
| 130 |
+
market = load_ohlcv("SPY", "2018-01-01", "2022-01-01", source="auto")
|
| 131 |
+
assert market.source == "synthetic"
|
| 132 |
+
assert not market.is_real
|
| 133 |
+
assert "unavailable" in market.note
|
| 134 |
+
|
| 135 |
+
def test_normalise_handles_yahoo_style_frames(self):
|
| 136 |
+
index = pd.date_range("2020-01-01", periods=5, tz="UTC")
|
| 137 |
+
raw = pd.DataFrame(
|
| 138 |
+
{"Open": 1.0, "High": 2.0, "Low": 0.5, "Adj Close": 1.5, "Volume": 10},
|
| 139 |
+
index=index,
|
| 140 |
+
)
|
| 141 |
+
out = _normalise(raw)
|
| 142 |
+
assert list(out.columns) == ["open", "high", "low", "close", "volume"]
|
| 143 |
+
assert out.index.tz is None
|
| 144 |
+
|
| 145 |
+
def test_normalise_rejects_frames_with_no_price(self):
|
| 146 |
+
with pytest.raises(ValueError, match="missing required column"):
|
| 147 |
+
_normalise(pd.DataFrame({"volume": [1, 2, 3]}))
|
| 148 |
+
|
| 149 |
+
def test_too_short_a_range_is_rejected(self):
|
| 150 |
+
with pytest.raises(ValueError, match="too short"):
|
| 151 |
+
simulate_ohlcv("SPY", "2020-01-01", "2020-01-10")
|
| 152 |
+
|
| 153 |
+
|
| 154 |
+
class TestVerdict:
|
| 155 |
+
def test_strong_evidence_outranks_weak_evidence(self):
|
| 156 |
+
metrics = {"n_trades": 300, "sharpe": 1.2, "total_return": 0.8, "max_drawdown": -0.2}
|
| 157 |
+
benchmark = {"sharpe": 0.4}
|
| 158 |
+
strong = reality_score(metrics, benchmark, p_value=0.001, dsr=0.99, pbo=0.02,
|
| 159 |
+
wf_efficiency=0.9, wf_win_rate=1.0, cost_stress_ratio=0.9)
|
| 160 |
+
weak = reality_score(metrics, benchmark, p_value=0.45, dsr=0.10, pbo=0.55,
|
| 161 |
+
wf_efficiency=-0.2, wf_win_rate=0.2, cost_stress_ratio=0.1)
|
| 162 |
+
assert strong["score"] > 85 > weak["score"]
|
| 163 |
+
assert strong["grade"] == "A" and weak["grade"] == "F"
|
| 164 |
+
|
| 165 |
+
def test_score_is_always_inside_the_scale(self):
|
| 166 |
+
for p in (0.0, 0.5, 1.0):
|
| 167 |
+
for dsr in (0.0, 1.0):
|
| 168 |
+
out = reality_score(
|
| 169 |
+
{"n_trades": 100, "sharpe": 0.5, "total_return": 0.2, "max_drawdown": -0.1},
|
| 170 |
+
{"sharpe": 0.1}, p_value=p, dsr=dsr, pbo=0.2,
|
| 171 |
+
wf_efficiency=0.5, cost_stress_ratio=0.5,
|
| 172 |
+
)
|
| 173 |
+
assert 0.0 <= out["score"] <= 100.0
|
| 174 |
+
|
| 175 |
+
def test_losing_money_caps_the_score(self):
|
| 176 |
+
out = reality_score(
|
| 177 |
+
{"n_trades": 200, "sharpe": 0.3, "total_return": -0.4, "max_drawdown": -0.6},
|
| 178 |
+
{"sharpe": 0.5}, p_value=0.001, dsr=0.99, pbo=0.01,
|
| 179 |
+
wf_efficiency=1.0, cost_stress_ratio=1.0,
|
| 180 |
+
)
|
| 181 |
+
assert out["score"] <= 50
|
| 182 |
+
assert any("lost money" in f for f in out["flags"])
|
| 183 |
+
|
| 184 |
+
def test_too_few_trades_is_flagged_and_capped(self):
|
| 185 |
+
out = reality_score(
|
| 186 |
+
{"n_trades": 3, "sharpe": 2.5, "total_return": 1.0, "max_drawdown": -0.1},
|
| 187 |
+
{"sharpe": 0.3}, p_value=0.001, dsr=0.99, pbo=0.01,
|
| 188 |
+
wf_efficiency=1.0, cost_stress_ratio=1.0,
|
| 189 |
+
)
|
| 190 |
+
assert out["score"] <= 55
|
| 191 |
+
assert any("coin flips" in f for f in out["flags"])
|
| 192 |
+
|
| 193 |
+
def test_closet_indexing_is_called_out(self):
|
| 194 |
+
out = reality_score(
|
| 195 |
+
{"n_trades": 50, "sharpe": 0.6, "total_return": 0.5, "max_drawdown": -0.2},
|
| 196 |
+
{"sharpe": 0.6}, benchmark_correlation=0.99,
|
| 197 |
+
)
|
| 198 |
+
assert any("repackaged long position" in f for f in out["flags"])
|
| 199 |
+
|
| 200 |
+
|
| 201 |
+
class TestLabEndToEnd:
|
| 202 |
+
def test_full_pipeline_produces_a_complete_report(self, monkeypatch):
|
| 203 |
+
monkeypatch.setattr("algotrader.lab.load_ohlcv", lambda *a, **k: MarketData(
|
| 204 |
+
"SIM", simulate_ohlcv("SPY", "2016-01-01", "2023-01-01"), "synthetic", "1d", "test"
|
| 205 |
+
))
|
| 206 |
+
report = run_lab(LabConfig(strategy="sma_cross", n_permutations=25, wf_folds=3, grid_limit=12))
|
| 207 |
+
|
| 208 |
+
assert 0.0 <= report.verdict["score"] <= 100.0
|
| 209 |
+
assert report.verdict["grade"] in {"A", "B", "C", "D", "F"}
|
| 210 |
+
assert 0 < report.permutation.p_value <= 1
|
| 211 |
+
assert 0.0 <= report.dsr["dsr"] <= 1.0
|
| 212 |
+
assert report.trials["n"] > 1
|
| 213 |
+
assert len(report.backtest.equity) == len(report.market.df)
|
| 214 |
+
assert report.cost_stress["sharpe_3x"] <= report.cost_stress["sharpe_1x"] + 1e-9
|
| 215 |
+
|
| 216 |
+
def test_too_little_history_gives_a_readable_error(self, monkeypatch):
|
| 217 |
+
short = simulate_ohlcv("SPY", "2020-01-01", "2020-06-01")
|
| 218 |
+
monkeypatch.setattr(
|
| 219 |
+
"algotrader.lab.load_ohlcv",
|
| 220 |
+
lambda *a, **k: MarketData("SIM", short, "synthetic", "1d", "test"),
|
| 221 |
+
)
|
| 222 |
+
with pytest.raises(ValueError, match="Widen the date range"):
|
| 223 |
+
run_lab(LabConfig(n_permutations=0, wf_folds=2))
|
| 224 |
+
|
| 225 |
+
def test_arena_ranks_every_strategy_and_keeps_the_controls(self, monkeypatch):
|
| 226 |
+
monkeypatch.setattr("algotrader.lab.load_ohlcv", lambda *a, **k: MarketData(
|
| 227 |
+
"SIM", simulate_ohlcv("SPY", "2017-01-01", "2022-01-01"), "synthetic", "1d", "test"
|
| 228 |
+
))
|
| 229 |
+
table, market, curves = run_arena(LabConfig(), n_permutations=0)
|
| 230 |
+
|
| 231 |
+
assert len(table) == len(REGISTRY)
|
| 232 |
+
assert {"buy_and_hold", "coin_flip"} <= set(table["key"])
|
| 233 |
+
assert table["Evidence"].is_monotonic_decreasing
|
| 234 |
+
assert set(curves) == set(REGISTRY)
|
tests/test_v2_validation.py
ADDED
|
@@ -0,0 +1,179 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Statistical machinery.
|
| 2 |
+
|
| 3 |
+
These tests matter more than the engine's, because a validation suite that
|
| 4 |
+
always says "no edge" is as useless as one that always says "great edge". Each
|
| 5 |
+
class below checks both directions: it must reject noise *and* detect signal.
|
| 6 |
+
"""
|
| 7 |
+
|
| 8 |
+
from __future__ import annotations
|
| 9 |
+
|
| 10 |
+
import numpy as np
|
| 11 |
+
import pandas as pd
|
| 12 |
+
import pytest
|
| 13 |
+
|
| 14 |
+
from algotrader.data import simulate_ohlcv
|
| 15 |
+
from algotrader.strategies import get_strategy
|
| 16 |
+
from algotrader.validation.deflated_sharpe import (
|
| 17 |
+
deflated_sharpe_ratio,
|
| 18 |
+
expected_max_sharpe,
|
| 19 |
+
min_track_record_length,
|
| 20 |
+
probabilistic_sharpe_ratio,
|
| 21 |
+
)
|
| 22 |
+
from algotrader.validation.pbo import probability_of_backtest_overfitting
|
| 23 |
+
from algotrader.validation.permutation import permutation_test, permute_bars
|
| 24 |
+
from algotrader.validation.walkforward import walk_forward
|
| 25 |
+
|
| 26 |
+
|
| 27 |
+
def trending_market(n: int = 2200, phi: float = 0.35, seed: int = 5) -> pd.DataFrame:
|
| 28 |
+
"""A market with genuine, exploitable serial correlation."""
|
| 29 |
+
rng = np.random.default_rng(seed)
|
| 30 |
+
returns = np.zeros(n)
|
| 31 |
+
noise = rng.normal(0, 0.01, n)
|
| 32 |
+
for i in range(1, n):
|
| 33 |
+
returns[i] = phi * returns[i - 1] + noise[i]
|
| 34 |
+
close = 100 * np.exp(np.cumsum(returns))
|
| 35 |
+
index = pd.date_range("2012-01-01", periods=n, freq="B")
|
| 36 |
+
return pd.DataFrame(
|
| 37 |
+
{"open": close, "high": close * 1.004, "low": close * 0.996, "close": close, "volume": 1e6},
|
| 38 |
+
index=index,
|
| 39 |
+
)
|
| 40 |
+
|
| 41 |
+
|
| 42 |
+
class TestPermutationMechanics:
|
| 43 |
+
def test_shuffling_preserves_the_distribution_of_moves(self):
|
| 44 |
+
df = simulate_ohlcv("SPY", "2018-01-01", "2023-01-01")
|
| 45 |
+
shuffled = permute_bars(df, np.random.default_rng(0))
|
| 46 |
+
|
| 47 |
+
assert len(shuffled) == len(df)
|
| 48 |
+
assert shuffled.index.equals(df.index)
|
| 49 |
+
original = np.sort(np.log(df["close"] / df["open"]).to_numpy()[1:])
|
| 50 |
+
permuted = np.sort(np.log(shuffled["close"] / shuffled["open"]).to_numpy()[1:])
|
| 51 |
+
np.testing.assert_allclose(original, permuted, rtol=1e-9)
|
| 52 |
+
|
| 53 |
+
def test_shuffling_keeps_bars_internally_valid(self):
|
| 54 |
+
df = simulate_ohlcv("AAPL", "2019-01-01", "2023-01-01")
|
| 55 |
+
shuffled = permute_bars(df, np.random.default_rng(3))
|
| 56 |
+
assert (shuffled["high"] >= shuffled["low"]).all()
|
| 57 |
+
assert (shuffled["high"] >= shuffled["close"]).all()
|
| 58 |
+
assert (shuffled["low"] <= shuffled["close"]).all()
|
| 59 |
+
assert (shuffled["close"] > 0).all()
|
| 60 |
+
|
| 61 |
+
def test_shuffling_destroys_serial_correlation(self):
|
| 62 |
+
df = trending_market()
|
| 63 |
+
real = df["close"].pct_change().dropna().autocorr(1)
|
| 64 |
+
shuffled = permute_bars(df, np.random.default_rng(1))["close"].pct_change().dropna().autocorr(1)
|
| 65 |
+
assert real > 0.2
|
| 66 |
+
assert abs(shuffled) < 0.1
|
| 67 |
+
|
| 68 |
+
def test_block_mode_retains_some_structure(self):
|
| 69 |
+
df = trending_market()
|
| 70 |
+
blocked = permute_bars(df, np.random.default_rng(2), method="block", block=40)
|
| 71 |
+
assert blocked["close"].pct_change().dropna().autocorr(1) > 0.1
|
| 72 |
+
|
| 73 |
+
def test_p_value_can_never_be_zero(self):
|
| 74 |
+
"""+1 correction: the observed run is itself a draw from the null."""
|
| 75 |
+
df = trending_market()
|
| 76 |
+
strategy = get_strategy("momentum")
|
| 77 |
+
result = permutation_test(
|
| 78 |
+
df, lambda f: strategy.generate(f, {"lookback": 5}), n_permutations=30, seed=0
|
| 79 |
+
)
|
| 80 |
+
assert result.p_value >= 1 / 31
|
| 81 |
+
assert 0 < result.p_value <= 1
|
| 82 |
+
|
| 83 |
+
|
| 84 |
+
class TestPermutationPower:
|
| 85 |
+
def test_real_edge_is_detected(self):
|
| 86 |
+
df = trending_market()
|
| 87 |
+
strategy = get_strategy("momentum")
|
| 88 |
+
result = permutation_test(
|
| 89 |
+
df, lambda f: strategy.generate(f, {"lookback": 5}), n_permutations=200, seed=1
|
| 90 |
+
)
|
| 91 |
+
assert result.observed > result.null.mean()
|
| 92 |
+
assert result.p_value < 0.05
|
| 93 |
+
|
| 94 |
+
def test_random_strategy_on_a_structureless_market_is_not_significant(self):
|
| 95 |
+
df = simulate_ohlcv("SIM", "2010-01-01", "2023-01-01")
|
| 96 |
+
strategy = get_strategy("coin_flip")
|
| 97 |
+
result = permutation_test(
|
| 98 |
+
df, lambda f: strategy.generate(f, {"hold": 5, "seed": 7}), n_permutations=200, seed=2
|
| 99 |
+
)
|
| 100 |
+
assert result.p_value > 0.05
|
| 101 |
+
|
| 102 |
+
|
| 103 |
+
class TestDeflatedSharpe:
|
| 104 |
+
def test_selection_bar_rises_with_the_number_of_trials(self):
|
| 105 |
+
low = expected_max_sharpe(10, 0.01)
|
| 106 |
+
high = expected_max_sharpe(1000, 0.01)
|
| 107 |
+
assert 0 < low < high
|
| 108 |
+
|
| 109 |
+
def test_a_single_trial_has_no_selection_bar(self):
|
| 110 |
+
assert expected_max_sharpe(1, 0.01) == 0.0
|
| 111 |
+
|
| 112 |
+
def test_more_trials_lowers_the_deflated_sharpe(self):
|
| 113 |
+
rng = np.random.default_rng(7)
|
| 114 |
+
returns = rng.normal(0.0006, 0.01, 2000)
|
| 115 |
+
few = deflated_sharpe_ratio(returns, 1.0, 252, n_trials=2, variance_of_trials=0.01)
|
| 116 |
+
many = deflated_sharpe_ratio(returns, 1.0, 252, n_trials=500, variance_of_trials=0.01)
|
| 117 |
+
assert many["dsr"] < few["dsr"]
|
| 118 |
+
assert many["psr"] == pytest.approx(few["psr"]) # PSR ignores selection
|
| 119 |
+
|
| 120 |
+
def test_psr_rises_with_track_record_length(self):
|
| 121 |
+
short = probabilistic_sharpe_ratio(0.05, 100)
|
| 122 |
+
long = probabilistic_sharpe_ratio(0.05, 5000)
|
| 123 |
+
assert 0.5 < short < long < 1.0
|
| 124 |
+
|
| 125 |
+
def test_negative_skew_and_fat_tails_are_penalised(self):
|
| 126 |
+
clean = probabilistic_sharpe_ratio(0.06, 1000, skew=0.0, kurtosis=3.0)
|
| 127 |
+
nasty = probabilistic_sharpe_ratio(0.06, 1000, skew=-1.5, kurtosis=12.0)
|
| 128 |
+
assert nasty < clean
|
| 129 |
+
|
| 130 |
+
def test_track_record_requirement_is_infinite_below_the_bar(self):
|
| 131 |
+
assert min_track_record_length(0.01, 500, benchmark=0.05) == float("inf")
|
| 132 |
+
assert np.isfinite(min_track_record_length(0.10, 500, benchmark=0.02))
|
| 133 |
+
|
| 134 |
+
|
| 135 |
+
class TestPBO:
|
| 136 |
+
def test_pure_noise_scores_near_one_half(self):
|
| 137 |
+
rng = np.random.default_rng(11)
|
| 138 |
+
matrix = rng.normal(0, 0.01, size=(1200, 30)) # 30 skill-free variants
|
| 139 |
+
result = probability_of_backtest_overfitting(matrix, n_splits=8)
|
| 140 |
+
assert 0.3 < result["pbo"] < 0.7
|
| 141 |
+
|
| 142 |
+
def test_a_genuinely_better_variant_is_not_flagged(self):
|
| 143 |
+
rng = np.random.default_rng(12)
|
| 144 |
+
matrix = rng.normal(0, 0.01, size=(1200, 20))
|
| 145 |
+
matrix[:, 3] += 0.004 # column 3 has a persistent, real edge
|
| 146 |
+
result = probability_of_backtest_overfitting(matrix, n_splits=8)
|
| 147 |
+
assert result["pbo"] < 0.15
|
| 148 |
+
assert result["most_selected_index"] == 3
|
| 149 |
+
assert result["selection_stability"] > 0.9
|
| 150 |
+
|
| 151 |
+
def test_too_few_variants_returns_nan_not_a_crash(self):
|
| 152 |
+
rng = np.random.default_rng(13)
|
| 153 |
+
result = probability_of_backtest_overfitting(rng.normal(0, 0.01, size=(500, 1)))
|
| 154 |
+
assert np.isnan(result["pbo"])
|
| 155 |
+
assert result["note"]
|
| 156 |
+
|
| 157 |
+
def test_odd_split_counts_are_made_even(self):
|
| 158 |
+
rng = np.random.default_rng(14)
|
| 159 |
+
result = probability_of_backtest_overfitting(rng.normal(0, 0.01, (800, 10)), n_splits=7)
|
| 160 |
+
assert result["n_combinations"] > 0
|
| 161 |
+
|
| 162 |
+
|
| 163 |
+
class TestWalkForward:
|
| 164 |
+
def test_a_real_edge_survives_out_of_sample(self):
|
| 165 |
+
result = walk_forward(trending_market(), get_strategy("momentum"), n_folds=4)
|
| 166 |
+
assert result["folds"]
|
| 167 |
+
assert result["mean_oos_sharpe"] > 0
|
| 168 |
+
assert result["efficiency"] > 0.3
|
| 169 |
+
|
| 170 |
+
def test_folds_do_not_overlap_train_and_test(self):
|
| 171 |
+
result = walk_forward(trending_market(), get_strategy("sma_cross"), n_folds=4)
|
| 172 |
+
for fold in result["folds"]:
|
| 173 |
+
assert fold["train_end"] <= fold["test_start"]
|
| 174 |
+
|
| 175 |
+
def test_short_history_degrades_gracefully(self):
|
| 176 |
+
df = trending_market(n=150)
|
| 177 |
+
result = walk_forward(df, get_strategy("momentum"), n_folds=5)
|
| 178 |
+
assert result["folds"] == []
|
| 179 |
+
assert result["note"]
|
tests/test_yahoo_data_stream.py
CHANGED
|
@@ -1,3 +1,5 @@
|
|
|
|
|
|
|
|
| 1 |
import pandas as pd
|
| 2 |
import pytest
|
| 3 |
from unittest.mock import patch
|
|
@@ -32,6 +34,39 @@ def _sample_yahoo_frame():
|
|
| 32 |
)
|
| 33 |
|
| 34 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 35 |
class TestYahooDataStream:
|
| 36 |
def test_initialization_from_symbol(self, yahoo_config):
|
| 37 |
stream = YahooDataStream(yahoo_config)
|
|
@@ -58,3 +93,137 @@ class TestYahooDataStream:
|
|
| 58 |
df = stream.get_historical_data('AAPL', '2024-01-01', '2024-12-31')
|
| 59 |
assert len(df) == 3
|
| 60 |
assert 'open' in df.columns
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import logging
|
| 2 |
+
|
| 3 |
import pandas as pd
|
| 4 |
import pytest
|
| 5 |
from unittest.mock import patch
|
|
|
|
| 34 |
)
|
| 35 |
|
| 36 |
|
| 37 |
+
def _split_frame():
|
| 38 |
+
"""An unadjusted 10:1 split, as Yahoo returns it with auto_adjust=False.
|
| 39 |
+
|
| 40 |
+
Modelled on NVDA, 10 June 2024: the raw Close drops from ~1200 to ~120 and
|
| 41 |
+
a backtest reads it as a -90% day.
|
| 42 |
+
"""
|
| 43 |
+
idx = pd.date_range('2024-06-06', periods=4, freq='D', tz='America/New_York')
|
| 44 |
+
close = [1200.0, 1208.0, 120.5, 121.0]
|
| 45 |
+
return pd.DataFrame(
|
| 46 |
+
{
|
| 47 |
+
'Open': close,
|
| 48 |
+
'High': [c * 1.01 for c in close],
|
| 49 |
+
'Low': [c * 0.99 for c in close],
|
| 50 |
+
'Close': close,
|
| 51 |
+
'Volume': [1_000_000] * 4,
|
| 52 |
+
},
|
| 53 |
+
index=idx,
|
| 54 |
+
)
|
| 55 |
+
|
| 56 |
+
|
| 57 |
+
def _dated_frame(timestamps, close=100.0):
|
| 58 |
+
return pd.DataFrame(
|
| 59 |
+
{
|
| 60 |
+
'timestamp': [pd.Timestamp(t) for t in timestamps],
|
| 61 |
+
'open': close,
|
| 62 |
+
'high': close,
|
| 63 |
+
'low': close,
|
| 64 |
+
'close': close,
|
| 65 |
+
'volume': 1_000.0,
|
| 66 |
+
}
|
| 67 |
+
)
|
| 68 |
+
|
| 69 |
+
|
| 70 |
class TestYahooDataStream:
|
| 71 |
def test_initialization_from_symbol(self, yahoo_config):
|
| 72 |
stream = YahooDataStream(yahoo_config)
|
|
|
|
| 93 |
df = stream.get_historical_data('AAPL', '2024-01-01', '2024-12-31')
|
| 94 |
assert len(df) == 3
|
| 95 |
assert 'open' in df.columns
|
| 96 |
+
|
| 97 |
+
|
| 98 |
+
class TestPriceAdjustment:
|
| 99 |
+
"""Unadjusted prices turn every split into a phantom crash."""
|
| 100 |
+
|
| 101 |
+
def test_adjustment_is_on_by_default(self):
|
| 102 |
+
stream = YahooDataStream({'trading': {'symbol': 'AAPL', 'timeframe': '1d'}})
|
| 103 |
+
assert stream.auto_adjust is True
|
| 104 |
+
|
| 105 |
+
def test_auto_adjust_is_passed_through_to_yfinance(self):
|
| 106 |
+
stream = YahooDataStream({'trading': {'symbol': 'AAPL', 'timeframe': '1d'}})
|
| 107 |
+
with patch('yfinance.download', return_value=_sample_yahoo_frame()) as download:
|
| 108 |
+
stream._download('AAPL', period='5d', interval='1d')
|
| 109 |
+
assert download.call_args.kwargs['auto_adjust'] is True
|
| 110 |
+
|
| 111 |
+
def test_opting_out_of_adjustment_warns(self, caplog):
|
| 112 |
+
with caplog.at_level(logging.WARNING):
|
| 113 |
+
YahooDataStream({
|
| 114 |
+
'trading': {'symbol': 'AAPL', 'timeframe': '1d'},
|
| 115 |
+
'yahoo': {'auto_adjust': False},
|
| 116 |
+
})
|
| 117 |
+
assert any('not split' in r.message.lower() for r in caplog.records)
|
| 118 |
+
|
| 119 |
+
def test_split_sized_move_is_flagged(self, yahoo_config, caplog):
|
| 120 |
+
stream = YahooDataStream(yahoo_config)
|
| 121 |
+
df = stream._normalize_ohlcv(_split_frame())
|
| 122 |
+
with caplog.at_level(logging.WARNING):
|
| 123 |
+
found = stream._warn_if_unadjusted('NVDA', df)
|
| 124 |
+
assert found == 1
|
| 125 |
+
assert any('auto_adjust' in r.message for r in caplog.records)
|
| 126 |
+
|
| 127 |
+
def test_ordinary_moves_are_not_flagged(self, yahoo_config, caplog):
|
| 128 |
+
stream = YahooDataStream(yahoo_config)
|
| 129 |
+
df = stream._normalize_ohlcv(_sample_yahoo_frame())
|
| 130 |
+
with caplog.at_level(logging.WARNING):
|
| 131 |
+
assert stream._warn_if_unadjusted('AAPL', df) == 0
|
| 132 |
+
|
| 133 |
+
def test_intraday_bars_are_not_split_checked(self, yahoo_config):
|
| 134 |
+
"""A 40% move in one minute is a halt or a fat finger, not a split."""
|
| 135 |
+
yahoo_config['trading']['timeframe'] = '1m'
|
| 136 |
+
stream = YahooDataStream(yahoo_config)
|
| 137 |
+
df = stream._normalize_ohlcv(_split_frame())
|
| 138 |
+
assert stream._warn_if_unadjusted('NVDA', df) == 0
|
| 139 |
+
|
| 140 |
+
|
| 141 |
+
class TestIncompleteBars:
|
| 142 |
+
"""The bar Yahoo is still building must not be reported as final."""
|
| 143 |
+
|
| 144 |
+
def test_forming_bar_is_dropped(self, yahoo_config):
|
| 145 |
+
stream = YahooDataStream(yahoo_config)
|
| 146 |
+
now = pd.Timestamp.now(tz='UTC').tz_convert(None).normalize()
|
| 147 |
+
df = _dated_frame([now - pd.Timedelta(days=2), now - pd.Timedelta(days=1), now])
|
| 148 |
+
kept = stream._drop_incomplete(df)
|
| 149 |
+
assert len(kept) == 2
|
| 150 |
+
assert kept['timestamp'].max() < now
|
| 151 |
+
|
| 152 |
+
def test_finished_bars_all_survive(self, yahoo_config):
|
| 153 |
+
stream = YahooDataStream(yahoo_config)
|
| 154 |
+
now = pd.Timestamp.now(tz='UTC').tz_convert(None).normalize()
|
| 155 |
+
df = _dated_frame([now - pd.Timedelta(days=5), now - pd.Timedelta(days=4)])
|
| 156 |
+
assert len(stream._drop_incomplete(df)) == 2
|
| 157 |
+
|
| 158 |
+
def test_opting_in_keeps_the_forming_bar(self, yahoo_config):
|
| 159 |
+
yahoo_config['yahoo']['emit_incomplete_bars'] = True
|
| 160 |
+
stream = YahooDataStream(yahoo_config)
|
| 161 |
+
now = pd.Timestamp.now(tz='UTC').tz_convert(None).normalize()
|
| 162 |
+
df = _dated_frame([now - pd.Timedelta(days=1), now])
|
| 163 |
+
assert len(stream._drop_incomplete(df)) == 2
|
| 164 |
+
|
| 165 |
+
def test_partial_bar_is_never_emitted_then_stranded(self, yahoo_config):
|
| 166 |
+
"""The bug this guards: emitting the forming bar advanced the watermark,
|
| 167 |
+
so the finished version of that same bar never reached a callback."""
|
| 168 |
+
stream = YahooDataStream(yahoo_config)
|
| 169 |
+
received = []
|
| 170 |
+
stream.add_data_callback(lambda kind, bar: received.append(bar))
|
| 171 |
+
|
| 172 |
+
now = pd.Timestamp.now(tz='UTC').tz_convert(None).normalize()
|
| 173 |
+
yesterday, today = now - pd.Timedelta(days=1), now
|
| 174 |
+
|
| 175 |
+
stream._ingest_new_bars('AAPL', _dated_frame([yesterday, today], close=100.0))
|
| 176 |
+
assert len(received) == 1 # only yesterday's completed bar
|
| 177 |
+
|
| 178 |
+
# Next day: what was the forming bar is now final and must arrive.
|
| 179 |
+
with patch.object(stream, '_drop_incomplete', side_effect=lambda d: d):
|
| 180 |
+
stream._ingest_new_bars('AAPL', _dated_frame([yesterday, today], close=105.0))
|
| 181 |
+
assert len(received) == 2
|
| 182 |
+
assert received[-1]['close'] == 105.0
|
| 183 |
+
|
| 184 |
+
|
| 185 |
+
class TestPollBackoff:
|
| 186 |
+
"""Yahoo rate-limits hard, and a fixed interval keeps you throttled."""
|
| 187 |
+
|
| 188 |
+
def test_success_polls_at_the_configured_interval(self, yahoo_config):
|
| 189 |
+
yahoo_config['yahoo']['poll_interval_seconds'] = 60
|
| 190 |
+
stream = YahooDataStream(yahoo_config)
|
| 191 |
+
stream._consecutive_failures = 0
|
| 192 |
+
assert 48 <= stream._next_delay() <= 72 # 60s +/- jitter
|
| 193 |
+
|
| 194 |
+
def test_delay_grows_with_consecutive_failures(self, yahoo_config):
|
| 195 |
+
yahoo_config['yahoo']['poll_interval_seconds'] = 60
|
| 196 |
+
stream = YahooDataStream(yahoo_config)
|
| 197 |
+
delays = []
|
| 198 |
+
for failures in (1, 2, 3):
|
| 199 |
+
stream._consecutive_failures = failures
|
| 200 |
+
delays.append(stream._next_delay())
|
| 201 |
+
assert delays[0] < delays[1] < delays[2]
|
| 202 |
+
|
| 203 |
+
def test_backoff_is_capped(self, yahoo_config):
|
| 204 |
+
yahoo_config['yahoo']['poll_interval_seconds'] = 60
|
| 205 |
+
yahoo_config['yahoo']['max_backoff_seconds'] = 300
|
| 206 |
+
stream = YahooDataStream(yahoo_config)
|
| 207 |
+
stream._consecutive_failures = 20
|
| 208 |
+
assert stream._next_delay() <= 300 * 1.2
|
| 209 |
+
|
| 210 |
+
def test_jitter_desynchronises_retries(self, yahoo_config):
|
| 211 |
+
stream = YahooDataStream(yahoo_config)
|
| 212 |
+
stream._consecutive_failures = 3
|
| 213 |
+
assert len({stream._next_delay() for _ in range(20)}) > 1
|
| 214 |
+
|
| 215 |
+
def test_poll_reports_failure_when_every_symbol_fails(self, yahoo_config):
|
| 216 |
+
stream = YahooDataStream(yahoo_config)
|
| 217 |
+
with patch.object(stream, '_download', side_effect=RuntimeError('429 Too Many Requests')):
|
| 218 |
+
assert stream._poll_once() is False
|
| 219 |
+
|
| 220 |
+
def test_poll_reports_success_when_a_symbol_returns_bars(self, yahoo_config):
|
| 221 |
+
stream = YahooDataStream(yahoo_config)
|
| 222 |
+
with patch.object(stream, '_download', return_value=_sample_yahoo_frame()):
|
| 223 |
+
assert stream._poll_once() is True
|
| 224 |
+
|
| 225 |
+
def test_empty_response_counts_as_failure(self, yahoo_config):
|
| 226 |
+
"""A rate-limited yfinance returns an empty frame rather than raising."""
|
| 227 |
+
stream = YahooDataStream(yahoo_config)
|
| 228 |
+
with patch.object(stream, '_download', return_value=pd.DataFrame()):
|
| 229 |
+
assert stream._poll_once() is False
|
ui/dash_app.py
CHANGED
|
@@ -158,11 +158,12 @@ class TradingDashApp:
|
|
| 158 |
dbc.Select(
|
| 159 |
id="data-source-select",
|
| 160 |
options=[
|
|
|
|
| 161 |
{"label": "CSV File", "value": "csv"},
|
| 162 |
{"label": "Alpaca API", "value": "alpaca"},
|
| 163 |
{"label": "Synthetic Data", "value": "synthetic"}
|
| 164 |
],
|
| 165 |
-
value="
|
| 166 |
)
|
| 167 |
], width=4),
|
| 168 |
dbc.Col([
|
|
|
|
| 158 |
dbc.Select(
|
| 159 |
id="data-source-select",
|
| 160 |
options=[
|
| 161 |
+
{"label": "Yahoo Finance", "value": "yahoo"},
|
| 162 |
{"label": "CSV File", "value": "csv"},
|
| 163 |
{"label": "Alpaca API", "value": "alpaca"},
|
| 164 |
{"label": "Synthetic Data", "value": "synthetic"}
|
| 165 |
],
|
| 166 |
+
value="yahoo"
|
| 167 |
)
|
| 168 |
], width=4),
|
| 169 |
dbc.Col([
|
ui/jupyter_widgets.py
CHANGED
|
@@ -62,8 +62,8 @@ class TradingJupyterUI:
|
|
| 62 |
|
| 63 |
# Data widgets
|
| 64 |
self.data_source = widgets.Dropdown(
|
| 65 |
-
options=['csv', 'alpaca', 'synthetic'],
|
| 66 |
-
value='
|
| 67 |
description='Data Source:',
|
| 68 |
style={'description_width': '120px'}
|
| 69 |
)
|
|
@@ -76,7 +76,7 @@ class TradingJupyterUI:
|
|
| 76 |
|
| 77 |
self.timeframe_input = widgets.Dropdown(
|
| 78 |
options=['1m', '5m', '15m', '1h', '1d'],
|
| 79 |
-
value='
|
| 80 |
description='Timeframe:',
|
| 81 |
style={'description_width': '120px'}
|
| 82 |
)
|
|
|
|
| 62 |
|
| 63 |
# Data widgets
|
| 64 |
self.data_source = widgets.Dropdown(
|
| 65 |
+
options=['yahoo', 'csv', 'alpaca', 'synthetic'],
|
| 66 |
+
value='yahoo',
|
| 67 |
description='Data Source:',
|
| 68 |
style={'description_width': '120px'}
|
| 69 |
)
|
|
|
|
| 76 |
|
| 77 |
self.timeframe_input = widgets.Dropdown(
|
| 78 |
options=['1m', '5m', '15m', '1h', '1d'],
|
| 79 |
+
value='1d',
|
| 80 |
description='Timeframe:',
|
| 81 |
style={'description_width': '120px'}
|
| 82 |
)
|